足球赛事爬虫返回NaN值/空DataFrame的原因及解决方法
问题
以下是用于爬取Oddsportal足球赛事数据的Python代码:
import os import time import threading import pandas as pd import numpy as np from math import nan from datetime import datetime, timedelta from multiprocessing.pool import ThreadPool from bs4 import BeautifulSoup as bs from selenium import webdriver from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.wait import WebDriverWait from selenium.webdriver.common.by import By from selenium.common.exceptions import TimeoutException import re class Driver: def __init__(self): options = webdriver.ChromeOptions() # options.add_argument("--headless") # Un-comment next line to supress logging: options.add_experimental_option('excludeSwitches', ['enable-logging']) self.driver = webdriver.Chrome(options=options) def __del__(self): self.driver.quit() # clean up driver when we are cleaned up threadLocal = threading.local() def create_driver(): the_driver = getattr(threadLocal, 'the_driver', None) if the_driver is None: the_driver = Driver() setattr(threadLocal, 'the_driver', the_driver) return the_driver.driver class GameData: def __init__(self): self.date = [] self.time = [] self.game = [] self.score = [] self.home_odds = [] self.draw_odds = [] self.away_odds = [] self.country = [] self.league = [] def generate_matches(pgSoup, defaultVal=None): evtSel = { 'time': 'div.next-m\\:flex-col p', 'game': 'a[title]', 'score': 'a[title]~div:not(.hidden)', 'home_odds': 'div[class^="flex-center flex-col gap-1 border-l border-black-ma"]:nth-child(2)', 'draw_odds': 'div[class^="flex-center flex-col gap-1 border-l border-black-ma"]:nth-child(3)', 'away_odds': 'div[class^="flex-center flex-col gap-1 border-l border-black-ma"]:nth-child(4)' } events, current_group = [], {} pgDate = pgSoup.select_one('h1.title[id="next-matches-h1"]') if pgDate: pgDate = pgDate.get_text().split(',', 1)[-1].strip() for evt in pgSoup.select('div[set]>div:last-child'): if evt.parent.select(f':scope>div:first-child+div+div'): cgVals = [v.get_text(' ').strip() if v else defaultVal for v in [ evt.parent.select_one(s) for s in [':scope>div:first-child+div>div:first-child', ':scope>div:first-child>a:nth-of-type(2):nth-last-of-type(2)', ':scope>div:first-child>a:nth-of-type(3):last-of-type']]] current_group = dict(zip(['date', 'country', 'league'], cgVals)) if pgDate: current_group['date'] = pgDate evtRow = {'date': current_group.get('date', defaultVal)} for k, v in evtSel.items(): v = evt.select_one(v).get_text(' ') if evt.select_one(v) else defaultVal evtRow[k] = ' '.join(v.split()) if isinstance(v, str) else v # evtTeams = evt.select('a div>a[title]') evtTeams = evt.select('a[title]') evtRow['game'] = ' - '.join(a['title'] for a in evtTeams) evtRow['country'] = current_group.get('country', defaultVal) evtRow['league'] = current_group.get('league', defaultVal) events.append(evtRow) return events def parse_data(url, return_urls=False): print(f'Parsing URL: {url}\n') browser = create_driver() browser.get(url) # Wait for the first match element to be present wait = WebDriverWait(browser, 20) match_element_class = "border-black-borders.group.flex" # Class of the match element wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, f".{match_element_class}"))) # Now, check if the first match element contains match data first_match_element = browser.find_element(By.CSS_SELECTOR, f".{match_element_class}") match_data_indicator = "participant-name" # Class indicating match data wait.until(EC.presence_of_element_located((By.CLASS_NAME, match_data_indicator))) # ########## For page to scroll to the end ########### scroll_pause_time = 4 try: # ########## For page to scroll to the end ########### scroll_pause_time = 4 # Get scroll height last_height = browser.execute_script("return document.body.scrollHeight") while True: # Scroll down to bottom browser.execute_script("window.scrollTo(0, document.body.scrollHeight);") # Wait to load page time.sleep(scroll_pause_time) # Calculate new scroll height and compare with last scroll height new_height = browser.execute_script("return document.body.scrollHeight") if new_height == last_height: break last_height = new_height # ########## For page to scroll to the end ########### # If browser stalls and gives timeout exception refresh the browser except TimeoutException: print("Timeout exception occurred, refreshing the page...") browser.refresh() time.sleep(5) soup = bs(browser.page_source, "lxml") game_data = GameData() game_keys = [a for a, av in game_data.__dict__.items() if isinstance(av, list)] for row in generate_matches(soup, defaultVal=nan): for k in game_keys: getattr(game_data, k).append(row.get(k, nan)) if return_urls: ac_sel = 'div:has(>a.active-item-calendar)' # a_cont selector a_sel = f'{ac_sel}>a[href]:not([href^="#"]):not(.active-item-calendar)' a_tags = soup.select(a_sel) if a_tags: urls = ['https://www.oddsportal.com' + a_tag['href'] for a_tag in a_tags] # print(f'urls after initial creation: {urls}') # Extract the date from the first URL last_date_str = urls[0].split('/')[-2] print(f'last date str: {last_date_str}') last_date = datetime.strptime(last_date_str, '%Y%m%d') # Generate the additional URLs for i in range(1, 4): new_date = last_date - timedelta(days=i) new_date_str = new_date.strftime('%Y%m%d') print(f'new dates: {new_date_str}') new_url = f'https://www.oddsportal.com/matches/football/{new_date_str}/' urls.append(new_url) # print(f'urls after generating additional URL #{i}: {urls}') else: urls = [] # print(f'final urls: {urls}') if urls and urls[-1].startswith('https://www.oddsportal.com/matches/football/'): # Extract the date from the last URL last_date_str = urls[0].split('/')[-2] print(last_date_str) else: print('No valid URLs found') return game_data, urls return game_data if __name__ == '__main__': games = None pool = ThreadPool(5) # Get today's data and the Urls for the other days: url_today = 'https://www.oddsportal.com/matches/soccer' game_data_today, urls = pool.apply(parse_data, args=(url_today, True)) game_data_results = pool.imap(parse_data, urls) # ########################### BUILD DATAFRAME ############################ game_data_dfList, added_todayGame = [], False for game_data in game_data_results: try: game_data_dfList.append(pd.DataFrame(game_data.__dict__)) if not added_todayGame: game_data_dfList += [pd.DataFrame(game_data_today.__dict__)] added_todayGame = True except Exception as e: game_n = len(game_data_dfList) + 1 print(f'Error tabulating game_data_df#{game_n}:\n{repr(e)}') try: games = pd.concat(game_data_dfList, ignore_index=True) except Exception as e: print('Error concatenating DataFrames:', repr(e)) # ######################################################################### print('!?NO GAMES?!' if games is None else games) # ensure all the drivers are "quitted": del threadLocal # a little extra insurance import gc gc.collect() games.to_csv()
该代码原本应输出包含date、time、game等字段的结构化表格,示例格式如下:
date | time | game | score | home_odds | draw_odds | away_odds | country | league -------------+--------+--------------------------------------+---------+-------------+-------------+-------------+-----------+---------- 01 Apr 2023 | 01:00 | Widad Adabi de Boufarik – Temouchent | 1 – 1 | 2.38 | 2.93 | 3.06 | Algeria | Ligue 2 01 Apr 2023 | 01:00 | Relizane – Oued Sly | 1 – 2 | 10.02 | 5 | 1.28 | Algeria | Ligue 2
但实际运行后得到空DataFrame,所有字段值均为NaN:
Empty DataFrame Columns: [date, time, game, score, home_odds, draw_odds, away_odds, country, league] Index: []
请问出现该问题的原因是什么?如何修改代码以获取符合要求的赛事数据?
原因分析与修复方案
核心原因
Oddsportal网站的页面结构已更新,原代码中的所有CSS选择器均已失效,无法定位到赛事数据相关元素,导致提取不到任何数据,最终生成空DataFrame。
修复步骤与修改代码
1. 更新CSS选择器
完全替换generate_matches函数中的选择器,适配当前页面结构,确保能正确抓取日期、国家、联赛、赛事时间、对阵双方、比分和赔率。
2. 优化数据提取逻辑
增加元素存在性判断,避免因部分元素缺失报错,同时将赔率转换为数值类型。
3. 修正DataFrame合并逻辑
调整数据合并顺序,确保今日数据被正确添加,最后去除全空行。
修改后的完整代码
import os import time import threading import pandas as pd import numpy as np from math import nan from datetime import datetime, timedelta from multiprocessing.pool import ThreadPool from bs4 import BeautifulSoup as bs from selenium import webdriver from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.wait import WebDriverWait from selenium.webdriver.common.by import By from selenium.common.exceptions import TimeoutException import re class Driver: def __init__(self): options = webdriver.ChromeOptions() # options.add_argument("--headless") # Un-comment next line to supress logging: options.add_experimental_option('excludeSwitches', ['enable-logging']) self.driver = webdriver.Chrome(options=options) def __del__(self): self.driver.quit() # clean up driver when we are cleaned up threadLocal = threading.local() def create_driver(): the_driver = getattr(threadLocal, 'the_driver', None) if the_driver is None: the_driver = Driver() setattr(threadLocal, 'the_driver', the_driver) return the_driver.driver class GameData: def __init__(self): self.date = [] self.time = [] self.game = [] self.score = [] self.home_odds = [] self.draw_odds = [] self.away_odds = [] self.country = [] self.league = [] def generate_matches(pgSoup, defaultVal=None): # 更新后的CSS选择器,适配当前Oddsportal页面结构 events, current_group = [], {} # 提取页面顶部的日期(如果存在) pg_date_elem = pgSoup.select_one('h1[class*="title"]') pg_date = pg_date_elem.get_text(strip=True).split(',')[-1].strip() if pg_date_elem else None # 遍历所有赛事分组(每个分组包含日期、国家、联赛及下属赛事) for group in pgSoup.select('div[class*="sport-content"] div[class*="event-group"]'): # 提取分组的日期、国家、联赛信息 date_elem = group.select_one('div[class*="event-group-header"] div[class*="date"]') country_elem = group.select_one('div[class*="event-group-header"] a[href*="/country/"]') league_elem = group.select_one('div[class*="event-group-header"] a[href*="/league/"]') current_group = { 'date': date_elem.get_text(strip=True) if date_elem else pg_date or defaultVal, 'country': country_elem.get_text(strip=True) if country_elem else defaultVal, 'league': league_elem.get_text(strip=True) if league_elem else defaultVal } # 遍历分组内的所有赛事 for match in group.select('div[class*="event-row"]'): evt_row = { 'date': current_group['date'], 'country': current_group['country'], 'league': current_group['league'] } # 提取赛事时间 time_elem = match.select_one('div[class*="time"]') evt_row['time'] = time_elem.get_text(strip=True) if time_elem else defaultVal # 提取对阵双方 teams = match.select('div[class*="participant"]') if len(teams) >= 2: evt_row['game'] = f"{teams[0].get_text(strip=True)} – {teams[1].get_text(strip=True)}" else: evt_row['game'] = defaultVal # 提取比分 score_elem = match.select_one('div[class*="score"]') evt_row['score'] = score_elem.get_text(strip=True) if score_elem else defaultVal # 提取赔率(主胜、平局、客胜) odds = match.select('div[class*="odds"] span[class*="odd"]') if len(odds) >= 3: evt_row['home_odds'] = float(odds[0].get_text(strip=True)) if odds[0].get_text(strip=True) else defaultVal evt_row['draw_odds'] = float(odds[1].get_text(strip=True)) if odds[1].get_text(strip=True) else defaultVal evt_row['away_odds'] = float(odds[2].get_text(strip=True)) if odds[2].get_text(strip=True) else defaultVal else: evt_row['home_odds'] = defaultVal evt_row['draw_odds'] = defaultVal evt_row['away_odds'] = defaultVal events.append(evt_row) return events def parse_data(url, return_urls=False): print(f'Parsing URL: {url}\n') browser = create_driver() browser.get(url) # 等待赛事列表加载完成 wait = WebDriverWait(browser, 20) try: wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, 'div[class*="event-row"]'))) except TimeoutException: print("Timeout loading matches, refreshing page...") browser.refresh() wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, 'div[class*="event-row"]'))) # 滚动加载所有赛事 scroll_pause_time = 3 last_height = browser.execute_script("return document.body.scrollHeight") while True: browser.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(scroll_pause_time) new_height = browser.execute_script("return document.body.scrollHeight") if new_height == last_height: break last_height = new_height time.sleep(2) soup = bs(browser.page_source, "lxml") game_data = GameData() game_keys = [k for k, v in game_data.__dict__.items() if isinstance(v, list)] for row in generate_matches(soup, defaultVal=nan):
相关产品推荐
相关产品推荐

