You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

BeautifulSoup find_all触发IndexError: list index out of range问题求助

问题描述

我正在做一个从足球网站爬取数据的个人项目,碰到了棘手问题:在get_fixtures_data函数的for循环末尾添加break语句时,代码无报错;但移除break后就触发IndexError,提示该函数中的results列表为空。我尝试用PyCharm调试查找列表为空的节点,但未成功定位。附上代码及报错信息,请求协助排查。

代码片段
from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.chrome.options import Options
from datetime import datetime
import pandas as pd


class FootballPredictions:

    LEAGUES = ["https://www.flashscore.co.za/soccer/england/premier-league/fixtures/",
               "https://www.flashscore.co.za/soccer/spain/laliga/fixtures/",
               "https://www.flashscore.co.za/soccer/germany/bundesliga/fixtures/",
               "https://www.flashscore.co.za/soccer/italy/serie-a/fixtures/",
               "https://www.flashscore.co.za/soccer/brazil/serie-a/fixtures/",
               "https://www.flashscore.co.za/soccer/france/ligue-1/fixtures/",
               "https://www.flashscore.co.za/soccer/portugal/liga-portugal/fixtures/",
               "https://www.flashscore.co.za/soccer/mexico/liga-mx/fixtures/",
               "https://www.flashscore.co.za/soccer/argentina/liga-profesional/fixtures/",
               "https://www.flashscore.co.za/soccer/turkey/super-lig/fixtures/",
               ]

    @classmethod
    def get_fixtures_ids(cls):
        options = Options()
        options.add_argument("--headless=new")
        browser = webdriver.Chrome(r"C:\WebDrivers\chromedriver.exe", options=options)

        day = datetime.today().day
        month = datetime.today().month
        # date = f"{day}.{month}."
        date = "01.04."

        for link in FootballPredictions.LEAGUES:
            browser.get(link)
            markup = browser.page_source

            soup = BeautifulSoup(markup, "html.parser")
            fixtures_divs = soup.find_all("div", attrs={"class": "event__match event__match--static " +
                                                                 "event__match--scheduled event__match--twoLine"})

            for fixture in fixtures_divs:
                fixture_date = fixture.find_next("div", {"class": "event__time"}).text.split(" ")[0]
                fixture_id = fixture["id"].split("_")[-1]

                if date == fixture_date:
                    with open("ids.txt", "a") as file:
                        file.write(fixture_id + "\n")

        browser.quit()

    @classmethod
    def get_fixtures_data(cls):
        options = Options()
        options.add_argument("--headless=new")
        browser = webdriver.Chrome(r"C:\WebDrivers\chromedriver.exe", options=options)

        with open("ids.txt", "r") as file:
            fixtures_identities = file.read().split("\n")[0:-1]

            for identity in fixtures_identities:
                fixture_identity = identity
                link = fr"https://www.flashscore.co.za/match/{fixture_identity}/#/h2h/overall"

                browser.get(link)
                markup = browser.page_source

                soup = BeautifulSoup(markup, "html.parser")

                teams = soup.find_all("a", {"class": "participant__participantName participant__overflow"})

                home_team = teams[0].text
                home_team_goals = 0

                away_team = teams[1].text
                away_team_goals = 0

                head_to_head_goals = 0

                results = soup.find_all("div", {"class": "h2h__section section"})

                home_team_stats = results[0].find_all("div", {"class": "h2h__row"})
                away_team_stats = results[1].find_all("div", {"class": "h2h__row"})
                head_to_head_stats = results[2].find_all("div", {"class": "h2h__row"})

                # home team row
                for row in home_team_stats:
                    name = row.find("span", {"class": "h2h__homeParticipant"}).find("span", {"h2h__participantInner"})

                    if name.text == home_team:
                        goal = row.find("span", {"class": "h2h__result"}).find_all("span")[0].text
                        home_team_goals += int(goal)

                    else:
                        goal = row.find("span", {"class": "h2h__result"}).find_all("span")[1].text
                        home_team_goals += int(goal)

                # away team row
                for row in away_team_stats:
                    name = row.find("span", {"class": "h2h__homeParticipant"}).find("span", {"h2h__participantInner"})

                    if name.text == away_team:
                        goal = row.find("span", {"class": "h2h__result"}).find_all("span")[0].text
                        away_team_goals += int(goal)

                    else:
                        goal = row.find("span", {"class": "h2h__result"}).find_all("span")[1].text
                        away_team_goals += int(goal)

                # head-to-head row
                for row in head_to_head_stats:
                    home_goals = row.find("span", {"class": "h2h__result"}).find_all("span")[0].text
                    away_goals = row.find("span", {"class": "h2h__result"}).find_all("span")[1].text
                    goals = int(home_goals) + int(away_goals)
                    head_to_head_goals += goals

                # write data to file
                with open("fixtures_data.txt", "a") as data_file:
                    data_file.write(f"{home_team},{away_team},{home_team_goals},{away_team_goals},{head_to_head_goals}"
                                    + "\n")


    @classmethod
    def clear_files(cls):
        with open("ids.txt", "w") as file:
            pass

        with open("fixtures_data.txt", "w") as file:
            pass

        with open("predictions.txt", "w") as file:
            pass

    @classmethod
    def run(cls):
        # FootballPredictions.get_fixtures_ids()
        FootballPredictions.get_fixtures_data()
        # FootballPredictions.clear_files()


FootballPredictions.run()
报错信息
Traceback (most recent call last):
  File "C:\Users\user\PycharmProjects\FootballPredictions\main.py", line 81, in get_fixtures_data
    home_team_stats = results[0].find_all("div", {"class": "h2h__row"})
                      ~~~~~~~^^^
IndexError: list index out of range
解决方案
  • 解决页面加载不完整问题
    无头模式下页面可能未完全渲染就被抓取,导致results为空。可以用Selenium的显式等待,确保目标元素加载完成后再获取源码:

    # 需要导入以下模块
    from selenium.webdriver.support.ui import WebDriverWait
    from selenium.webdriver.support import expected_conditions as EC
    from selenium.webdriver.common.by import By
    
    # 修改get_fixtures_data函数中browser.get(link)后的代码
    browser.get(link)
    # 等待最多10秒,直到h2h区块出现
    WebDriverWait(browser, 10).until(
        EC.presence_of_element_located((By.CLASS_NAME, "h2h__section"))
    )
    markup = browser.page_source
    
  • 添加空列表判断,避免索引越界
    在访问results的索引前,先检查列表是否为空,直接跳过无效的赛事ID:

    results = soup.find_all("div", {"class": "h2h__section section"})
    if not results:
        print(f"赛事ID {identity} 无匹配数据,跳过")
        continue
    # 后续代码正常执行
    home_team_stats = results[0].find_all("div", {"class": "h2h__row"})
    ...
    
  • 排查无效赛事ID
    打印当前处理的赛事ID,方便定位是哪个ID导致页面无数据:

    for identity in fixtures_identities:
        print(f"正在处理赛事ID: {identity}")
        fixture_identity = identity
        ...
    
  • 优化请求间隔,避免反爬限制
    频繁请求可能触发网站反爬机制,导致页面加载异常。可在循环中加入短暂间隔:

    import time
    
    # 在每次browser.get(link)后添加间隔
    browser.get(link)
    time.sleep(1)  # 间隔1秒,降低请求频率
    

内容的提问来源于stack exchange,提问作者user19792349

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.26 12:32:22