You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

足球赛事爬虫返回NaN值/空DataFrame的原因及解决方法

问题

以下是用于爬取Oddsportal足球赛事数据的Python代码:

import os
import time
import threading
import pandas as pd
import numpy as np
from math import nan
from datetime import datetime, timedelta
from multiprocessing.pool import ThreadPool
from bs4 import BeautifulSoup as bs
from selenium import webdriver
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.common.exceptions import TimeoutException
import re


class Driver:
    def __init__(self):
        options = webdriver.ChromeOptions()
        # options.add_argument("--headless")
        # Un-comment next line to supress logging:
        options.add_experimental_option('excludeSwitches', ['enable-logging'])
        self.driver = webdriver.Chrome(options=options)

    def __del__(self):
        self.driver.quit()  # clean up driver when we are cleaned up


threadLocal = threading.local()


def create_driver():
    the_driver = getattr(threadLocal, 'the_driver', None)
    if the_driver is None:
        the_driver = Driver()
        setattr(threadLocal, 'the_driver', the_driver)
    return the_driver.driver


class GameData:
    def __init__(self):
        self.date = []
        self.time = []
        self.game = []
        self.score = []
        self.home_odds = []
        self.draw_odds = []
        self.away_odds = []
        self.country = []
        self.league = []


def generate_matches(pgSoup, defaultVal=None):
    evtSel = {
        'time': 'div.next-m\\:flex-col p',
        'game': 'a[title]',
        'score': 'a[title]~div:not(.hidden)',
        'home_odds': 'div[class^="flex-center flex-col gap-1 border-l border-black-ma"]:nth-child(2)',
        'draw_odds': 'div[class^="flex-center flex-col gap-1 border-l border-black-ma"]:nth-child(3)',
        'away_odds': 'div[class^="flex-center flex-col gap-1 border-l border-black-ma"]:nth-child(4)'
    }

    events, current_group = [], {}
    pgDate = pgSoup.select_one('h1.title[id="next-matches-h1"]')
    if pgDate: pgDate = pgDate.get_text().split(',', 1)[-1].strip()
    for evt in pgSoup.select('div[set]>div:last-child'):
        if evt.parent.select(f':scope>div:first-child+div+div'):
            cgVals = [v.get_text(' ').strip() if v else defaultVal for v in [
                evt.parent.select_one(s) for s in
                [':scope>div:first-child+div>div:first-child',
                 ':scope>div:first-child>a:nth-of-type(2):nth-last-of-type(2)',
                 ':scope>div:first-child>a:nth-of-type(3):last-of-type']]]
            current_group = dict(zip(['date', 'country', 'league'], cgVals))
            if pgDate: current_group['date'] = pgDate

        evtRow = {'date': current_group.get('date', defaultVal)}

        for k, v in evtSel.items():
            v = evt.select_one(v).get_text(' ') if evt.select_one(v) else defaultVal
            evtRow[k] = ' '.join(v.split()) if isinstance(v, str) else v
        # evtTeams = evt.select('a div>a[title]')
        evtTeams = evt.select('a[title]')
        evtRow['game'] = ' - '.join(a['title'] for a in evtTeams)
        evtRow['country'] = current_group.get('country', defaultVal)
        evtRow['league'] = current_group.get('league', defaultVal)

        events.append(evtRow)

    return events


def parse_data(url, return_urls=False):
    print(f'Parsing URL: {url}\n')
    browser = create_driver()
    browser.get(url)
    # Wait for the first match element to be present
    wait = WebDriverWait(browser, 20)
    match_element_class = "border-black-borders.group.flex"  # Class of the match element
    wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, f".{match_element_class}")))

    # Now, check if the first match element contains match data
    first_match_element = browser.find_element(By.CSS_SELECTOR, f".{match_element_class}")
    match_data_indicator = "participant-name"  # Class indicating match data
    wait.until(EC.presence_of_element_located((By.CLASS_NAME, match_data_indicator)))
    
    # ########## For page to scroll to the end ###########
    scroll_pause_time = 4
    try:
        # ########## For page to scroll to the end ###########
        scroll_pause_time = 4

        # Get scroll height
        last_height = browser.execute_script("return document.body.scrollHeight")

        while True:
            # Scroll down to bottom
            browser.execute_script("window.scrollTo(0, document.body.scrollHeight);")

            # Wait to load page
            time.sleep(scroll_pause_time)

            # Calculate new scroll height and compare with last scroll height
            new_height = browser.execute_script("return document.body.scrollHeight")
            if new_height == last_height:
                break
            last_height = new_height
        # ########## For page to scroll to the end ###########
        # If browser stalls and gives timeout exception refresh the browser
    except TimeoutException:
        print("Timeout exception occurred, refreshing the page...")
        browser.refresh()

    time.sleep(5)
    soup = bs(browser.page_source, "lxml")

    game_data = GameData()
    game_keys = [a for a, av in game_data.__dict__.items() if isinstance(av, list)]
    for row in generate_matches(soup, defaultVal=nan):
        for k in game_keys: getattr(game_data, k).append(row.get(k, nan))
    if return_urls:
        ac_sel = 'div:has(>a.active-item-calendar)'  # a_cont selector
        a_sel = f'{ac_sel}>a[href]:not([href^="#"]):not(.active-item-calendar)'
        a_tags = soup.select(a_sel)

        if a_tags:
            urls = ['https://www.oddsportal.com' + a_tag['href'] for a_tag in a_tags]
            # print(f'urls after initial creation: {urls}')

            # Extract the date from the first URL
            last_date_str = urls[0].split('/')[-2]
            print(f'last date str: {last_date_str}')
            last_date = datetime.strptime(last_date_str, '%Y%m%d')

            # Generate the additional URLs
            for i in range(1, 4):
                new_date = last_date - timedelta(days=i)
                new_date_str = new_date.strftime('%Y%m%d')
                print(f'new dates: {new_date_str}')
                new_url = f'https://www.oddsportal.com/matches/football/{new_date_str}/'
                urls.append(new_url)
                # print(f'urls after generating additional URL #{i}: {urls}')
        else:
            urls = []

        # print(f'final urls: {urls}')

        if urls and urls[-1].startswith('https://www.oddsportal.com/matches/football/'):
            # Extract the date from the last URL
            last_date_str = urls[0].split('/')[-2]
            print(last_date_str)
        else:
            print('No valid URLs found')
        return game_data, urls
    return game_data


if __name__ == '__main__':
    games = None
    pool = ThreadPool(5)
    # Get today's data and the Urls for the other days:
    url_today = 'https://www.oddsportal.com/matches/soccer'
    game_data_today, urls = pool.apply(parse_data, args=(url_today, True))
    game_data_results = pool.imap(parse_data, urls)

    # ########################### BUILD  DATAFRAME ############################
    game_data_dfList, added_todayGame = [], False
    for game_data in game_data_results:
        try:
            game_data_dfList.append(pd.DataFrame(game_data.__dict__))
            if not added_todayGame:
                game_data_dfList += [pd.DataFrame(game_data_today.__dict__)]
                added_todayGame = True
        except Exception as e:
            game_n = len(game_data_dfList) + 1
            print(f'Error tabulating game_data_df#{game_n}:\n{repr(e)}')
    try:
        games = pd.concat(game_data_dfList, ignore_index=True)
    except Exception as e:
        print('Error concatenating DataFrames:', repr(e))
    # #########################################################################
    print('!?NO GAMES?!' if games is None else games)
    # ensure all the drivers are "quitted":
    del threadLocal  # a little extra insurance
    import gc

    gc.collect()

games.to_csv()    

该代码原本应输出包含date、time、game等字段的结构化表格,示例格式如下:

date        | time   | game                                 | score   |   home_odds |   draw_odds |   away_odds | country   | league
-------------+--------+--------------------------------------+---------+-------------+-------------+-------------+-----------+----------
 01 Apr 2023 | 01:00  | Widad Adabi de Boufarik – Temouchent | 1 – 1   |        2.38 |        2.93 |        3.06 | Algeria   | Ligue 2
 01 Apr 2023 | 01:00  | Relizane – Oued Sly                  | 1 – 2   |       10.02 |        5    |        1.28 | Algeria   | Ligue 2

但实际运行后得到空DataFrame,所有字段值均为NaN:

Empty DataFrame
Columns: [date, time, game, score, home_odds, draw_odds, away_odds, country, league]
Index: []

请问出现该问题的原因是什么?如何修改代码以获取符合要求的赛事数据?


原因分析与修复方案

核心原因

Oddsportal网站的页面结构已更新,原代码中的所有CSS选择器均已失效,无法定位到赛事数据相关元素,导致提取不到任何数据,最终生成空DataFrame。

修复步骤与修改代码

1. 更新CSS选择器

完全替换generate_matches函数中的选择器,适配当前页面结构,确保能正确抓取日期、国家、联赛、赛事时间、对阵双方、比分和赔率。

2. 优化数据提取逻辑

增加元素存在性判断,避免因部分元素缺失报错,同时将赔率转换为数值类型。

3. 修正DataFrame合并逻辑

调整数据合并顺序,确保今日数据被正确添加,最后去除全空行。

修改后的完整代码

import os
import time
import threading
import pandas as pd
import numpy as np
from math import nan
from datetime import datetime, timedelta
from multiprocessing.pool import ThreadPool
from bs4 import BeautifulSoup as bs
from selenium import webdriver
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.common.exceptions import TimeoutException
import re


class Driver:
    def __init__(self):
        options = webdriver.ChromeOptions()
        # options.add_argument("--headless")
        # Un-comment next line to supress logging:
        options.add_experimental_option('excludeSwitches', ['enable-logging'])
        self.driver = webdriver.Chrome(options=options)

    def __del__(self):
        self.driver.quit()  # clean up driver when we are cleaned up


threadLocal = threading.local()


def create_driver():
    the_driver = getattr(threadLocal, 'the_driver', None)
    if the_driver is None:
        the_driver = Driver()
        setattr(threadLocal, 'the_driver', the_driver)
    return the_driver.driver


class GameData:
    def __init__(self):
        self.date = []
        self.time = []
        self.game = []
        self.score = []
        self.home_odds = []
        self.draw_odds = []
        self.away_odds = []
        self.country = []
        self.league = []


def generate_matches(pgSoup, defaultVal=None):
    # 更新后的CSS选择器,适配当前Oddsportal页面结构
    events, current_group = [], {}
    
    # 提取页面顶部的日期(如果存在)
    pg_date_elem = pgSoup.select_one('h1[class*="title"]')
    pg_date = pg_date_elem.get_text(strip=True).split(',')[-1].strip() if pg_date_elem else None

    # 遍历所有赛事分组(每个分组包含日期、国家、联赛及下属赛事)
    for group in pgSoup.select('div[class*="sport-content"] div[class*="event-group"]'):
        # 提取分组的日期、国家、联赛信息
        date_elem = group.select_one('div[class*="event-group-header"] div[class*="date"]')
        country_elem = group.select_one('div[class*="event-group-header"] a[href*="/country/"]')
        league_elem = group.select_one('div[class*="event-group-header"] a[href*="/league/"]')
        
        current_group = {
            'date': date_elem.get_text(strip=True) if date_elem else pg_date or defaultVal,
            'country': country_elem.get_text(strip=True) if country_elem else defaultVal,
            'league': league_elem.get_text(strip=True) if league_elem else defaultVal
        }

        # 遍历分组内的所有赛事
        for match in group.select('div[class*="event-row"]'):
            evt_row = {
                'date': current_group['date'],
                'country': current_group['country'],
                'league': current_group['league']
            }

            # 提取赛事时间
            time_elem = match.select_one('div[class*="time"]')
            evt_row['time'] = time_elem.get_text(strip=True) if time_elem else defaultVal

            # 提取对阵双方
            teams = match.select('div[class*="participant"]')
            if len(teams) >= 2:
                evt_row['game'] = f"{teams[0].get_text(strip=True)} – {teams[1].get_text(strip=True)}"
            else:
                evt_row['game'] = defaultVal

            # 提取比分
            score_elem = match.select_one('div[class*="score"]')
            evt_row['score'] = score_elem.get_text(strip=True) if score_elem else defaultVal

            # 提取赔率(主胜、平局、客胜)
            odds = match.select('div[class*="odds"] span[class*="odd"]')
            if len(odds) >= 3:
                evt_row['home_odds'] = float(odds[0].get_text(strip=True)) if odds[0].get_text(strip=True) else defaultVal
                evt_row['draw_odds'] = float(odds[1].get_text(strip=True)) if odds[1].get_text(strip=True) else defaultVal
                evt_row['away_odds'] = float(odds[2].get_text(strip=True)) if odds[2].get_text(strip=True) else defaultVal
            else:
                evt_row['home_odds'] = defaultVal
                evt_row['draw_odds'] = defaultVal
                evt_row['away_odds'] = defaultVal

            events.append(evt_row)

    return events


def parse_data(url, return_urls=False):
    print(f'Parsing URL: {url}\n')
    browser = create_driver()
    browser.get(url)
    
    # 等待赛事列表加载完成
    wait = WebDriverWait(browser, 20)
    try:
        wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, 'div[class*="event-row"]')))
    except TimeoutException:
        print("Timeout loading matches, refreshing page...")
        browser.refresh()
        wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, 'div[class*="event-row"]')))

    # 滚动加载所有赛事
    scroll_pause_time = 3
    last_height = browser.execute_script("return document.body.scrollHeight")
    while True:
        browser.execute_script("window.scrollTo(0, document.body.scrollHeight);")
        time.sleep(scroll_pause_time)
        new_height = browser.execute_script("return document.body.scrollHeight")
        if new_height == last_height:
            break
        last_height = new_height

    time.sleep(2)
    soup = bs(browser.page_source, "lxml")

    game_data = GameData()
    game_keys = [k for k, v in game_data.__dict__.items() if isinstance(v, list)]
    for row in generate_matches(soup, defaultVal=nan):
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.12 03:02:21