BeautifulSoup find_all触发IndexError: list index out of range问题求助
问题描述
我正在做一个从足球网站爬取数据的个人项目,碰到了棘手问题:在get_fixtures_data函数的for循环末尾添加break语句时,代码无报错;但移除break后就触发IndexError,提示该函数中的results列表为空。我尝试用PyCharm调试查找列表为空的节点,但未成功定位。附上代码及报错信息,请求协助排查。
代码片段
from bs4 import BeautifulSoup from selenium import webdriver from selenium.webdriver.chrome.options import Options from datetime import datetime import pandas as pd class FootballPredictions: LEAGUES = ["https://www.flashscore.co.za/soccer/england/premier-league/fixtures/", "https://www.flashscore.co.za/soccer/spain/laliga/fixtures/", "https://www.flashscore.co.za/soccer/germany/bundesliga/fixtures/", "https://www.flashscore.co.za/soccer/italy/serie-a/fixtures/", "https://www.flashscore.co.za/soccer/brazil/serie-a/fixtures/", "https://www.flashscore.co.za/soccer/france/ligue-1/fixtures/", "https://www.flashscore.co.za/soccer/portugal/liga-portugal/fixtures/", "https://www.flashscore.co.za/soccer/mexico/liga-mx/fixtures/", "https://www.flashscore.co.za/soccer/argentina/liga-profesional/fixtures/", "https://www.flashscore.co.za/soccer/turkey/super-lig/fixtures/", ] @classmethod def get_fixtures_ids(cls): options = Options() options.add_argument("--headless=new") browser = webdriver.Chrome(r"C:\WebDrivers\chromedriver.exe", options=options) day = datetime.today().day month = datetime.today().month # date = f"{day}.{month}." date = "01.04." for link in FootballPredictions.LEAGUES: browser.get(link) markup = browser.page_source soup = BeautifulSoup(markup, "html.parser") fixtures_divs = soup.find_all("div", attrs={"class": "event__match event__match--static " + "event__match--scheduled event__match--twoLine"}) for fixture in fixtures_divs: fixture_date = fixture.find_next("div", {"class": "event__time"}).text.split(" ")[0] fixture_id = fixture["id"].split("_")[-1] if date == fixture_date: with open("ids.txt", "a") as file: file.write(fixture_id + "\n") browser.quit() @classmethod def get_fixtures_data(cls): options = Options() options.add_argument("--headless=new") browser = webdriver.Chrome(r"C:\WebDrivers\chromedriver.exe", options=options) with open("ids.txt", "r") as file: fixtures_identities = file.read().split("\n")[0:-1] for identity in fixtures_identities: fixture_identity = identity link = fr"https://www.flashscore.co.za/match/{fixture_identity}/#/h2h/overall" browser.get(link) markup = browser.page_source soup = BeautifulSoup(markup, "html.parser") teams = soup.find_all("a", {"class": "participant__participantName participant__overflow"}) home_team = teams[0].text home_team_goals = 0 away_team = teams[1].text away_team_goals = 0 head_to_head_goals = 0 results = soup.find_all("div", {"class": "h2h__section section"}) home_team_stats = results[0].find_all("div", {"class": "h2h__row"}) away_team_stats = results[1].find_all("div", {"class": "h2h__row"}) head_to_head_stats = results[2].find_all("div", {"class": "h2h__row"}) # home team row for row in home_team_stats: name = row.find("span", {"class": "h2h__homeParticipant"}).find("span", {"h2h__participantInner"}) if name.text == home_team: goal = row.find("span", {"class": "h2h__result"}).find_all("span")[0].text home_team_goals += int(goal) else: goal = row.find("span", {"class": "h2h__result"}).find_all("span")[1].text home_team_goals += int(goal) # away team row for row in away_team_stats: name = row.find("span", {"class": "h2h__homeParticipant"}).find("span", {"h2h__participantInner"}) if name.text == away_team: goal = row.find("span", {"class": "h2h__result"}).find_all("span")[0].text away_team_goals += int(goal) else: goal = row.find("span", {"class": "h2h__result"}).find_all("span")[1].text away_team_goals += int(goal) # head-to-head row for row in head_to_head_stats: home_goals = row.find("span", {"class": "h2h__result"}).find_all("span")[0].text away_goals = row.find("span", {"class": "h2h__result"}).find_all("span")[1].text goals = int(home_goals) + int(away_goals) head_to_head_goals += goals # write data to file with open("fixtures_data.txt", "a") as data_file: data_file.write(f"{home_team},{away_team},{home_team_goals},{away_team_goals},{head_to_head_goals}" + "\n") @classmethod def clear_files(cls): with open("ids.txt", "w") as file: pass with open("fixtures_data.txt", "w") as file: pass with open("predictions.txt", "w") as file: pass @classmethod def run(cls): # FootballPredictions.get_fixtures_ids() FootballPredictions.get_fixtures_data() # FootballPredictions.clear_files() FootballPredictions.run()
报错信息
Traceback (most recent call last): File "C:\Users\user\PycharmProjects\FootballPredictions\main.py", line 81, in get_fixtures_data home_team_stats = results[0].find_all("div", {"class": "h2h__row"}) ~~~~~~~^^^ IndexError: list index out of range
解决方案
解决页面加载不完整问题
无头模式下页面可能未完全渲染就被抓取,导致results为空。可以用Selenium的显式等待,确保目标元素加载完成后再获取源码:# 需要导入以下模块 from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.common.by import By # 修改get_fixtures_data函数中browser.get(link)后的代码 browser.get(link) # 等待最多10秒,直到h2h区块出现 WebDriverWait(browser, 10).until( EC.presence_of_element_located((By.CLASS_NAME, "h2h__section")) ) markup = browser.page_source添加空列表判断,避免索引越界
在访问results的索引前,先检查列表是否为空,直接跳过无效的赛事ID:results = soup.find_all("div", {"class": "h2h__section section"}) if not results: print(f"赛事ID {identity} 无匹配数据,跳过") continue # 后续代码正常执行 home_team_stats = results[0].find_all("div", {"class": "h2h__row"}) ...排查无效赛事ID
打印当前处理的赛事ID,方便定位是哪个ID导致页面无数据:for identity in fixtures_identities: print(f"正在处理赛事ID: {identity}") fixture_identity = identity ...优化请求间隔,避免反爬限制
频繁请求可能触发网站反爬机制,导致页面加载异常。可在循环中加入短暂间隔:import time # 在每次browser.get(link)后添加间隔 browser.get(link) time.sleep(1) # 间隔1秒,降低请求频率
内容的提问来源于stack exchange,提问作者user19792349
相关产品推荐
相关产品推荐

