You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python爬取Oddsportal赔率数据报IndexError列表越界问题求解

问题修复与功能实现方案

一、IndexError报错修复

报错原因

原代码的国家/联赛提取逻辑是针对单个联赛详情页编写的,你当前使用的是全站足球赛事汇总页(/matches/soccer/xxx格式),这类页面的<th class="first2 tl">节点下的<a>标签数量不足3个,访问count[2]时直接触发下标越界。

修复逻辑

汇总页中每个赛事区块的头部会单独标注对应国家和联赛,我们调整提取逻辑,遍历每个赛事区块分别提取国家、联赛字段,同时增加长度校验避免越界。

二、分页爬取功能实现

实现步骤

  1. 先请求后续赛事首页,用你提供的XPath提取分页栏的最大页码
  2. 按照Oddsportal分页规则生成所有分页URL:https://www.oddsportal.com/matches/soccer/page/{页码}/
  3. 多线程并发爬取所有分页的赛事数据
  4. 合并所有分页的结果为单个DataFrame

完整修改后代码

import pandas as pd
from bs4 import BeautifulSoup as bs
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import threading
from multiprocessing.pool import ThreadPool
import time

class Driver:
    def __init__(self):
        options = webdriver.ChromeOptions()
        options.add_argument("--headless=new")
        options.add_argument("--disable-blink-features=AutomationControlled")
        # 取消注释关闭日志
        # options.add_experimental_option('excludeSwitches', ['enable-logging'])
        self.driver = webdriver.Chrome(options=options)
        self.driver.implicitly_wait(10) # 隐式等待10秒,等待页面元素加载

    def __del__(self):
        try:
            self.driver.quit()
        except:
            pass


threadLocal = threading.local()

def create_driver():
    the_driver = getattr(threadLocal, 'the_driver', None)
    if the_driver is None:
        the_driver = Driver()
        setattr(threadLocal, 'the_driver', the_driver)
    return the_driver.driver


class GameData:
    def __init__(self):
        self.date = []
        self.time = []
        self.game = []
        self.score = []
        self.home_odds = []
        self.draw_odds = []
        self.away_odds = []
        self.country = []
        self.league = []


def get_total_pages(base_url="https://www.oddsportal.com/matches/soccer/"):
    """获取总页数"""
    driver = create_driver()
    driver.get(base_url)
    try:
        # 等待分页元素加载
        WebDriverWait(driver, 10).until(
            EC.presence_of_element_located((By.XPATH, '//*[@id="col-content"]/div[3]/div/div/span'))
        )
        page_spans = driver.find_elements(By.XPATH, '//*[@id="col-content"]/div[3]/div/div/span')
        # 提取所有数字页码,取最大值
        page_nums = []
        for span in page_spans:
            text = span.text.strip()
            if text.isdigit():
                page_nums.append(int(text))
        return max(page_nums) if page_nums else 1
    except Exception as e:
        print(f"获取总页数失败: {e}")
        return 1


def parse_data(url):
    browser = create_driver()
    try:
        browser.get(url)
        # 等待赛事表格加载
        WebDriverWait(browser, 10).until(
            EC.presence_of_element_located((By.ID, 'tournamentTable'))
        )
        time.sleep(1) # 加1秒延迟避免反爬
        html = browser.page_source
        soup = bs(html, "lxml")
        table = soup.find('table', {'id': 'tournamentTable'})
        if not table:
            return None
        
        game_data = GameData()
        current_country = ""
        current_league = ""
        current_date = ""

        for row in table.find_all('tr'):
            # 处理国家+联赛行
            if 'center' in row.get('class', []) and row.find('th', class_='first2 tl'):
                a_tags = row.find('th', class_='first2 tl').find_all('a')
                if len(a_tags) >= 2:
                    current_country = a_tags[0].text.strip()
                    current_league = a_tags[1].text.strip()
                continue
            # 处理日期行
            if 'nob-border' in row.get('class', []):
                date_text = row.find('th').text.strip()
                current_date = date_text.split(' - ')[0] if ' - ' in date_text else date_text
                continue
            # 处理赛事行
            cols = row.find_all('td')
            if len(cols) < 7:
                continue
            time_str = cols[0].text.strip()
            if ':' not in time_str:
                continue
            game_data.date.append(current_date)
            game_data.time.append(time_str)
            game_data.game.append(cols[1].text.strip())
            game_data.score.append(cols[2].text.strip())
            game_data.home_odds.append(cols[3].text.strip())
            game_data.draw_odds.append(cols[4].text.strip())
            game_data.away_odds.append(cols[5].text.strip())
            game_data.country.append(current_country)
            game_data.league.append(current_league)
        return game_data
    except Exception as e:
        print(f"解析页面{url}失败: {e}")
        return None


if __name__ == '__main__':
    base_url = "https://www.oddsportal.com/matches/soccer/"
    # 获取总页数
    total_pages = get_total_pages(base_url)
    print(f"共检测到{total_pages}页后续赛事")
    # 生成所有分页URL
    urls = [base_url] + [f"{base_url}page/{i}/" for i in range(2, total_pages+1)]
    # 可以根据需求调整爬取页数,比如只爬前3页就改成 urls = urls[:3]
    
    results_list = []
    MAX_BROWSERS = 5 # 控制最大并发数,避免被封
    pool = ThreadPool(min(MAX_BROWSERS, len(urls)))
    for game_data in pool.imap(parse_data, urls):
        if game_data is None:
            continue
        df = pd.DataFrame(game_data.__dict__)
        results_list.append(df)
    
    # 合并所有结果
    if results_list:
        final_results = pd.concat(results_list, ignore_index=True)
        print(final_results.head())
        # 可以保存到csv
        # final_results.to_csv("oddsportal_football_matches.csv", index=False, encoding="utf_8_sig")
    else:
        print("未爬取到任何数据")
    
    # 清理资源
    del threadLocal
    import gc
    gc.collect()

注意事项

  • 需提前安装对应Chrome版本的ChromeDriver,或使用webdriver-manager自动管理驱动
  • 可以根据自己的需求调整MAX_BROWSERS数值,并发数太高容易触发反爬导致IP被封
  • 如果需要爬取历史日期的赛事,修改URL规则为https://www.oddsportal.com/matches/soccer/YYYYMMDD/即可

内容的提问来源于stack exchange,提问作者user16304089

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.05 11:33:01