You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

爬取fbref.com时出现Beautiful Soup列表索引越界错误如何解决?

问题

尝试爬取英超各球队过去几年比赛日志时,运行以下Web Scraping代码触发IndexError错误:

import requests

standings_url = "https://fbref.com/en/comps/9/Premier-League-Stats"
data = requests.get(standings_url)

from bs4 import BeautifulSoup
import pandas as pd
import time


soup = BeautifulSoup(data.text)
pd.set_option('display.max_columns', None)
pd.set_option('display.max_rows', None)
pd.set_option('display.expand_frame_repr', False)

standings_table = soup.select('table.stats_table')[0]

links = standings_table.find_all('a')
links = [l.get("href") for l in links]
links = [l for l in links if '/squads/' in l]

team_urls = [f"https://fbref.com{l}" for l in links]
data = requests.get(team_urls[0])

matches = pd.read_html(str(data.text), match="Scores & Fixtures")[0]

soup = BeautifulSoup(data.text)

links = soup.find_all('a')
links = [l.get("href") for l in links]
links = [l for l in links if l and 'all_comps/shooting/' in l]

data = requests.get(f"https://fbref.com{links[0]}")
shooting = pd.read_html(str(data.text), match="Shooting")[0]

shooting.columns = shooting.columns.droplevel()

team_data = matches.merge(shooting[["Date", "Sh", "SoT", "Dist", "FK", "PK", "PKatt"]], on="Date")

years = list(range(2023, 2020, -1))
all_matches = []

standings_url = "https://fbref.com/en/comps/9/Premier-League-Stats"

for year in years:
    data = requests.get(standings_url)
    soup = BeautifulSoup(data.text)
    standings_table = soup.select('table.stats_table')[0]

    links = [l.get("href") for l in standings_table.find_all('a')]
    links = [l for l in links if '/squads/' in l]
    team_urls = [f"https://fbref.com{l}" for l in links]

    previous_season = soup.select("a.prev")[0].get("href")
    standings_url = f"https://fbref.com{previous_season}"

    for team_url in team_urls:
        team_name = team_url.split('/')[-1].replace("-Stats", "").replace("-", " ")
        data = requests.get(team_url)
        matches = pd.read_html(str(data.text), match="Scores & Fixtures")[0]
        soup = BeautifulSoup(data.text)
        links = [l.get("href") for l in soup.find_all('a')]
        links = [l for l in links if l and 'all_comps/shooting/' in l]
        data = requests.get(f"https://fbref.com{links[0]}")
        shooting = pd.read_html(str(data.text), match="Shooting")[0]
        shooting.columns = shooting.columns.droplevel()
        try:
            team_data = matches.merge(shooting[["Date", "Sh", "SoT", "Dist", "FK", "PK", "PKatt"]], on="Date")
        except ValueError:
            continue

        team_data["Season"] = year
        team_data["Team"] = team_name
        all_matches.append(team_data)
        time.sleep(1)
        break

match_df = pd.concat(all_matches)
match_df.columns = [c.lower() for c in match_df.columns]

match_df.to_csv("matches.csv")
print(match_df)

错误信息:

Traceback (most recent call last):
  File "C:\Users\user\PycharmProjects\WebScrapingEPL\EPL webscrape.py", line 48, in <module>
    standings_table = soup.select('table.stats_table')[0]
IndexError: list index out of range

错误出现在standings_table = soup.select('table.stats_table')[0]行,提示列表索引越界。


解决思路与修复方案

核心原因

  1. 反爬拦截:fbref会检测请求头,无合理User-Agent的请求会被返回不含目标表格的页面,导致soup.select('table.stats_table')返回空列表。
  2. 无容错处理:直接通过[0]索引取列表元素,未判断列表是否为空,一旦爬取失败就触发索引越界。

修复步骤

  1. 添加请求头:给requests.get()添加模拟浏览器的User-Agent,避免被反爬拦截。
  2. 增加空值判断:在取索引前先检查列表是否有内容,避免索引越界。
  3. 优化异常处理:对请求和页面解析环节增加异常捕获,提升代码稳定性。

修复后的代码

import requests
from bs4 import BeautifulSoup
import pandas as pd
import time

# 模拟浏览器请求头,避免被反爬拦截
headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36'
}

pd.set_option('display.max_columns', None)
pd.set_option('display.max_rows', None)
pd.set_option('display.expand_frame_repr', False)

years = list(range(2023, 2020, -1))
all_matches = []
standings_url = "https://fbref.com/en/comps/9/Premier-League-Stats"

for year in years:
    try:
        data = requests.get(standings_url, headers=headers)
        data.raise_for_status()  # 检查请求是否成功
        soup = BeautifulSoup(data.text, 'html.parser')
        
        # 先判断是否找到目标表格
        standings_tables = soup.select('table.stats_table')
        if not standings_tables:
            print(f"未找到赛季{year}的积分榜表格,跳过")
            break
        
        standings_table = standings_tables[0]
        links = [l.get("href") for l in standings_table.find_all('a')]
        links = [l for l in links if '/squads/' in l]
        team_urls = [f"https://fbref.com{l}" for l in links]

        # 获取上一赛季链接时同样增加判断
        prev_links = soup.select("a.prev")
        if not prev_links:
            print("未找到上一赛季链接,结束循环")
            break
        previous_season = prev_links[0].get("href")
        standings_url = f"https://fbref.com{previous_season}"

        for team_url in team_urls:
            try:
                team_name = team_url.split('/')[-1].replace("-Stats", "").replace("-", " ")
                data = requests.get(team_url, headers=headers)
                data.raise_for_status()
                
                # 处理比赛数据
                matches_list = pd.read_html(data.text, match="Scores & Fixtures")
                if not matches_list:
                    print(f"{team_name}未找到比赛数据,跳过")
                    continue
                matches = matches_list[0]
                
                soup = BeautifulSoup(data.text, 'html.parser')
                links = [l.get("href") for l in soup.find_all('a')]
                links = [l for l in links if l and 'all_comps/shooting/' in l]
                
                if not links:
                    print(f"{team_name}未找到射门数据链接,跳过")
                    continue
                
                data = requests.get(f"https://fbref.com{links[0]}", headers=headers)
                data.raise_for_status()
                
                shooting_list = pd.read_html(data.text, match="Shooting")
                if not shooting_list:
                    print(f"{team_name}未找到射门数据,跳过")
                    continue
                shooting = shooting_list[0]
                shooting.columns = shooting.columns.droplevel()
                
                try:
                    team_data = matches.merge(shooting[["Date", "Sh", "SoT", "Dist", "FK", "PK", "PKatt"]], on="Date")
                except ValueError:
                    print(f"{team_name}比赛数据与射门数据无法合并,跳过")
                    continue

                team_data["Season"] = year
                team_data["Team"] = team_name
                all_matches.append(team_data)
                time.sleep(1)
                break  # 仅爬取每赛季第一支球队,可根据需求删除此句
                
            except Exception as e:
                print(f"处理{team_name}时出错: {str(e)}")
                time.sleep(2)
                continue
                
    except Exception as e:
        print(f"处理赛季{year}时出错: {str(e)}")
        time.sleep(2)
        continue

if all_matches:
    match_df = pd.concat(all_matches)
    match_df.columns = [c.lower() for c in match_df.columns]
    match_df.to_csv("matches.csv", index=False)
    print(match_df)
else:
    print("未爬取到任何数据")

额外说明

  • 代码中保留了break语句(爬取每赛季第一支球队后停止),如果需要爬取所有球队,删除该语句即可。
  • 增加了time.sleep()避免请求过于频繁,降低被封禁风险。
  • 所有请求环节都增加了raise_for_status(),可以及时捕获HTTP请求错误(如403、404)。

内容的提问来源于stack exchange,提问作者Pratyush Shankar

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.20 06:00:12