You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium-Python中Stale Element异常问题求助

Selenium-Python爬取HLTV选手数据时遭遇Stale Element Reference Exception

问题现象

  • 首次爬取选手数据完全正常,但循环处理下一位选手时触发Stale Element Reference Exception,甚至无法打印"Found playerCol element",首次迭代后while循环停滞
  • 此前脚本可正常处理5位选手,添加获胜回合与总回合统计逻辑后出现该bug,怀疑嵌套循环导致
  • 已尝试多次重新初始化player_stats变量,问题未解决

原代码

from selenium import webdriver
from selenium.webdriver.support.ui import WebDriverWait
from selenium.common.exceptions import NoSuchElementException
from selenium.common.exceptions import StaleElementReferenceException
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.by import By
import pandas as pd
import re
import time

# Initialize the webdriver
driver = webdriver.Firefox()

# Navigate to the website
url = "https://www.hltv.org/stats/players"
driver.get(url)

WebDriverWait(driver, 15).until(EC.element_to_be_clickable((By.ID, "CybotCookiebotDialogBodyLevelButtonLevelOptinAllowAll"))).click()

# Find the elements containing the player statistics
player_stats = WebDriverWait(driver, 10).until(
    EC.presence_of_all_elements_located((By.CSS_SELECTOR, ".playerCol, .statsDetail"))
)


# Extract the relevant data from the elements
players = []

for i, player_stat in enumerate(player_stats):
    try:
        WebDriverWait(driver, 10).until(EC.presence_of_element_located((By.CSS_SELECTOR, ".playerCol, .statsDetail")))
        while True:
            player_stats = WebDriverWait(driver, 10).until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, ".playerCol, .statsDetail")))
            try:    
                if "playerCol" in player_stat.get_attribute("class"):
                    print("Found playerCol element")
                    name = player_stat.find_element(By.CSS_SELECTOR, "a").text if player_stat.find_elements(By.CSS_SELECTOR, "a") else player_stat.text
                    print(f"Name: {name}")
                elif "statsDetail" in player_stat.get_attribute("class"):
                    stats = player_stat.text.split()
                    if len(stats) >= 1 and re.search(r"\d+\.\d+", stats[0]):
                        kd_ratio = stats[0]
                break
            except StaleElementReferenceException as e:
                player_stats = WebDriverWait(driver, 10).until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, ".playerCol, .statsDetail")))
                player_stats = driver.find_elements(By.CSS_SELECTOR, ".playerCol, .statsDetail")
                print(f"An error occurred while processing match stats: {e}")
                break

        # Extract the player stats
        if "statsDetail" in player_stat.get_attribute("class"):
            stats = player_stat.text.split()
            if len(stats) >= 1 and re.search(r"\d+\.\d+", stats[0]):
                kd_ratio = stats[0]

                # Process match stats for the player
                try:
                    time.sleep(1)
                    WebDriverWait(driver, 15).until(EC.presence_of_element_located((By.CSS_SELECTOR, ".playerCol, .statsDetail")))
                    player_link = driver.find_element(By.XPATH, f"//a[contains(text(), '{name}')]")
                    print(player_link.get_attribute('outerHTML'))
                    driver.execute_script("arguments[0].click();", player_link)
                    time.sleep(1)
                    player_stats = driver.find_elements(By.CSS_SELECTOR, ".playerCol, .statsDetail")
                    player = [name, kd_ratio]

                    # Extract additional player stats
                    headshot_percentage = WebDriverWait(driver, 5).until(EC.presence_of_element_located((By.XPATH, "//span[contains(text(), 'Headshot %')]/following-sibling::span"))).text
                    player.append(headshot_percentage)

                    kpr = WebDriverWait(driver, 5).until(EC.presence_of_element_located((By.XPATH, "//span[contains(text(), 'Kills / round')]/following-sibling::span"))).text
                    player.append(kpr)

                    dpr = WebDriverWait(driver, 5).until(EC.presence_of_element_located((By.XPATH, "//span[contains(text(), 'Deaths / round')]/following-sibling::span"))).text
                    player.append(dpr)

                    # Extract match stats for the player
                    matches_link = WebDriverWait(driver, 5).until(EC.presence_of_element_located((By.CSS_SELECTOR, "a[href*='/stats/players/matches/'][data-link-tracking-destination='Click on Matches -> Individual -> Overview [subnavigation]']")))
                    driver.execute_script("arguments[0].click();", matches_link)
                    
                    match_stats = WebDriverWait(driver, 5).until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, "tr.group-2, tr.group-1")))
                    match_scores = []
                    num_of_matches = 0
                    rounds_won = 0
                    rounds_played = 0
                    # Process match stats for the player
                    for i, match_stat in enumerate(match_stats):
                        player_name = player[0]
                        player_team = driver.find_element(By.CSS_SELECTOR, ".gtSmartphone-only span:last-of-type").text
                        try:
                            team_name = ""
                            score = ""
                            while team_name == "" or score == "":
                                try:
                                    team = match_stat.find_element(By.CSS_SELECTOR, ".gtSmartphone-only span:last-of-type").text
                                    team_name = team.strip()
                                    
                                    score_span = match_stat.find_element(By.XPATH, ".//div[contains(@class, 'gtSmartphone-only')]//*[contains(text(), '(')]")
                                    score_text = score_span.text.strip()
                                
                                    score = re.search(r'\((\d+)\)', score_text).group(1)
                                    
                                except:
                                    time.sleep(1)
                                    match_stats = WebDriverWait(driver, 5).until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, "tr.group-2, tr.group-1")))
                                    match_stat = match_stats[i]
                            team_data = match_stat.find_elements(By.CSS_SELECTOR, ".gtSmartphone-only span")
                            print("Team data:", team_data[3].text)
                            if team_name.lower() == player_team.lower():
                                player_score = score
                                opposing_team_name = team_data[2].text.strip()
                                print(opposing_team_name)
                                opposing_team_score = team_data[3].text.strip('()')
                                print("Score strip: ", opposing_team_score)
                                rounds_won += int(player_score)
                                rounds_played += int(player_score) + int(opposing_team_score)
                            else:
                                player_score = team_data[1].text.strip('()')
                                print(player_score)
                                opposing_team_score = score
                                print(opposing_team_score)
                                opposing_team_name = team_data[0].text.strip()
                                print(opposing_team_name)
                                rounds_won += int(opposing_team_score)
                                rounds_played += int(player_score) + int(opposing_team_score)

                            match_scores.append((team_name, opposing_team_name, player_score, opposing_team_score))
                            num_of_matches += 1

                            if num_of_matches == 5: # exit loop after 5 iterations
                                break

                        except:
                            # Refresh the page if the element can't be found
                            driver.back()
                            player_stats = driver.find_elements(By.CSS_SELECTOR, ".playerCol, .statsDetail")
                            time.sleep(1)
                            match_stats = WebDriverWait(driver, 5).until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, "tr.group-2, tr.group-1")))

                except Exception as e:
                    print(f"An error occurred while processing data for player {name}: {e}")
                    continue

                players.append([name, kd_ratio, headshot_percentage, kpr, dpr, rounds_won, rounds_played])
                print(players)
                print(f"{player_name}: {rounds_won} rounds won out of {rounds_played} rounds played in {num_of_matches} matches")
                driver.get(url)
                time.sleep(1)
    except StaleElementReferenceException as e:
    # handle the exception here
        print(f"An error occurred while processing match stats: {e}")
        break
# Close the webdriver
driver.quit()
# Store the data in a Pandas dataframe
df = pd.DataFrame(players, columns=["Name", "K/D", "HS %", "KPR", "DPR", "RW", "RP"])

# Clean the data
df["K/D"] = df["K/D"].str.extract(r"(\d+\.\d+)").astype(float)
df["HS %"] = df["HS %"].str.extract(r"(\d+\.\d+)").astype(float)
df["KPR"] = df["KPR"].str.extract(r"(\d+\.\d+)").astype(float)
df["DPR"] = df["DPR"].str.extract(r"(\d+\.\d+)").astype(float)



# Drop any rows that have missing or invalid data
df.dropna(subset=["Name", "K/D", "HS %", "KPR", "DPR"], inplace=True)


# Save the data to a CSV file
df.to_csv("player_stats.csv", index=False, sep='\t')

# Close the webdriver
driver.quit() 

问题根源分析

  1. 元素引用失效:首次爬取后调用driver.get(url)回到选手列表页,最初获取的player_stats元素集合已失效(页面DOM重新渲染),后续循环仍操作旧元素引用,必然触发异常
  2. 循环逻辑混乱:外层for循环依赖初始元素集合,页面刷新后集合内元素全部过期,无法继续遍历
  3. 嵌套循环的元素复用问题:处理单选手数据时的嵌套循环未正确获取当前页面元素,而是复用旧引用

修复方案

1. 重构循环逻辑,避免依赖过期元素集合

不再基于初始player_stats做外层循环,改为每次回到列表页后重新获取选手列表,用计数器控制爬取数量:

2. 优化页面跳转后的元素等待

每次从详情页返回列表页后,必须等待列表元素重新加载完成,再进行下一次操作

3. 修复比赛数据解析逻辑

修正选手队伍信息的获取方式,改用更稳定的元素定位解析比分

完整修复代码

from selenium import webdriver
from selenium.webdriver.support.ui import WebDriverWait
from selenium.common.exceptions import StaleElementReferenceException
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.by import By
import pandas as pd
import re
import time

driver = webdriver.Firefox()
url = "https://www.hltv.org/stats/players"
driver.get(url)

# 同意Cookie
WebDriverWait(driver, 15).until(EC.element_to_be_clickable((By.ID, "CybotCookiebotDialogBodyLevelButtonLevelOptinAllowAll"))).click()

players = []
target_count = 5
current = 0

while current < target_count:
    try:
        # 每次回到列表页都重新获取选手元素
        player_links = WebDriverWait(driver, 10).until(
            EC.presence_of_all_elements_located((By.CSS_SELECTOR, ".playerCol a"))
        )
        if current >= len(player_links):
            print("页面已无更多选手")
            break
        
        # 获取当前选手基础信息
        player_link = player_links[current]
        name = player_link.text
        # 精准定位对应选手的KD值
        kd_ratio = WebDriverWait(driver, 5).until(
            EC.presence_of_element_located((By.XPATH, f"//a[text()='{name}']/ancestor::tr//td[contains(@class, 'statsDetail')][1]"))
        ).text
        
        # 进入选手详情页
        WebDriverWait(driver, 10).until(EC.element_to_be_clickable(player_link)).click()
        
        # 获取选手基础数据
        headshot_perc = WebDriverWait(driver, 5).until(
            EC.presence_of_element_located((By.XPATH, "//span[text()='Headshot %']/following-sibling::span"))
        ).text
        kpr = WebDriverWait(driver, 5).until(
            EC.presence_of_element_located((By.XPATH, "//span[text()='Kills / round']/following-sibling::span"))
        ).text
        dpr = WebDriverWait(driver, 5).until(
            EC.presence_of_element_located((By.XPATH, "//span[text()='Deaths / round']/following-sibling::span"))
        ).text
        # 从详情页获取选手当前队伍
        player_team = WebDriverWait(driver, 5).until(
            EC.presence_of_element_located((By.CSS_SELECTOR, ".playerTeam .text-ellipsis"))
        ).text
        
        # 进入比赛列表页
        matches_link = WebDriverWait(driver, 5).until(
            EC.element_to_be_clickable((By.CSS_SELECTOR, "a[href*='/stats/players/matches/']"))
        )
        driver.execute_script("arguments[0].click();", matches_link)
        
        # 处理前5场比赛数据
        rounds_won = 0
        rounds_played = 0
        matches = WebDriverWait(driver, 5).until(
            EC.presence_of_all_elements_located((By.CSS_SELECTOR, "tr.group-1, tr.group-2"))
        )
        
        for match in matches[:5]:
            try:
                # 解析比赛队伍与比分
                score_text = match.find_element(By.CSS_SELECTOR, ".gtSmartphone-only").text
                scores = re.findall(r'\((\d+)\)', score_text)
                if len(scores) != 2:
                    continue
                team_spans = match.find_elements(By.CSS_SELECTOR, ".gtSmartphone-only span")
                team1 = team_spans[0].text.strip()
                team2 = team_spans[2].text.strip()
                score1 = int(scores[0])
                score2 = int(scores[1])
                
                # 统计选手队伍的获胜回合
                if team1.lower() == player_team.lower():
                    rounds_won += score1
                elif team2.lower() == player_team.lower():
                    rounds_won += score2
                rounds_played += score1 + score2
                
            except Exception as e:
                print(f"处理{name}的比赛数据出错: {e}")
                continue
        
        # 保存当前选手数据
        players.append([name, kd_ratio, headshot_perc, kpr, dpr, rounds_won, rounds_played])
        print(f"完成{name}的数据爬取: {rounds_won}/{rounds_played} 回合")
        
        # 返回选手列表页,准备下一次爬取
        driver.get(url)
        current += 1
        
    except StaleElementReferenceException as e:
        print(f"元素过期,重新加载页面: {e}")
        driver.get(url)
        continue
    except Exception as e:
        print(f"通用错误: {e}")
        driver.get(url)
        continue

# 关闭浏览器并处理数据
driver.quit()
df = pd.DataFrame(players, columns=["Name", "K/D", "HS %", "KPR", "DPR", "RW", "RP"])
# 清洗数据
df["K/D"] = df["K/D"].str.extract(r"(\d+\.\d+)").astype(float)
df["HS %"] = df["HS %"].str.extract(r"(\d+\.\d+)").astype(float)
df["KPR"] = df["KPR"].str.extract(r"(\d+\.\d+)").astype(float)
df["DPR"] = df["DPR"].str.extract(r"(\d+\.\d+)").astype(float)
df.dropna(subset=["Name", "K/D", "HS %", "KPR", "DPR"], inplace=True)
df.to_csv("player_stats.csv", index=False, sep='\t')

关键修复点总结

  • 每次返回列表页都重新获取选手元素集合,彻底避免过期元素引用
  • 外层循环改为计数器+重新获取元素的模式,不再依赖初始DOM元素
  • 替换冗余的time.sleep为WebDriverWait显式等待,提升稳定性
  • 优化元素定位逻辑,使用更精准的选择器减少定位失败概率
  • 修正比赛数据解析逻辑,避免依赖不稳定的元素位置

内容的提问来源于stack exchange,提问作者Blue

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 11:20:28