You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium脚本仅爬取Forebet7场比赛数据的原因与解决方案

问题描述

我正在使用Selenium开展网页爬取项目,从足球预测网站(以EXAMPLE指代FOREBET)爬取足球比赛数据。但网页上明明列出了更多比赛,我的脚本却仅能获取7场比赛的数据。相关代码如下:

import time
from bs4 import BeautifulSoup
import pandas as pd
import logging
import google_colab_selenium as gs
from selenium.webdriver.chrome.options import Options
from random import choice

# Configurations
request_interval = 2  # seconds
page_load_delay = 2  # seconds
base_url = "https://www.example.com/en/football-tips-and-predictions-for-today/predictions-1x2"

# User agents list (non-mobile)
user_agents = [
    'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3',
    'Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/56.0.2924.87 Safari/537.36',
    'Mozilla/5.0 (Windows NT 6.1; WOW64; Trident/7.0; AS; rv:11.0) like Gecko',
    'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_6) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/14.0.1 Safari/605.1.15',
    'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/88.0.4324.96 Safari/537.36'
]

# Configure logging
logging.basicConfig(level=logging.INFO)

# Setup Chrome options and driver
def setup_selenium():
    options = Options()
    options.add_argument('--headless')
    options.add_argument('--no-sandbox')
    options.add_argument('--disable-dev-shm-usage')
    options.add_argument("--window-size=1920,1080")
    options.add_argument("--disable-infobars")
    options.add_argument("--disable-popup-blocking")
    options.add_argument("--ignore-certificate-errors")
    options.add_argument("--incognito")
    options.add_argument(f'--user-agent={choice(user_agents)}')
    driver = gs.Chrome(options=options)
    return driver

def get_page_content(url, driver, request_interval=2, page_load_delay=2):
    driver.get(url)
    time.sleep(request_interval)
    html_content = driver.page_source
    time.sleep(page_load_delay)
    return html_content

def parse_match_details(soup, driver):
    matches = []
    match_links = [a['href'] for a in soup.select('.contentmiddle a[itemprop="url"]')]
    for link in match_links:
        match_url = f"https://www.example.com{link}"
        html_content = get_page_content(match_url, driver)
        match_soup = BeautifulSoup(html_content, 'lxml')
        if not match_soup:
            continue

        try:
            home_team_name = match_soup.select_one("[itemprop='homeTeam'] [itemprop='url'] span").text.strip()
            away_team_name = match_soup.select_one("[itemprop='awayTeam'] [itemprop='url'] span").text.strip()
            home_matches_played = match_soup.select_one(".os_played_games_main span[data-team='h']").text.strip()
            away_matches_played = match_soup.select_one(".os_played_games_main span[data-team='a']").text.strip()
            home_standing = match_soup.select_one("span.teamtableleft").text.strip()
            away_standing = match_soup.select_one("span.teamtableright").text.strip()
            match_time = match_soup.select_one("[itemprop='startDate'] div").text.strip()

            recent_matches = match_soup.select_one("div.os_goals_section3_container")
            home_and_away_goals = recent_matches

            def get_statistic(selector, data_team, data_stat):
                stat = home_and_away_goals.select_one(f"[data-team='{data_team}'][data-stat='{data_stat}'] span.__{selector}")
                return stat.text.strip() if stat else ""

            home_stats = {
                "Under 1.5": get_statistic("under", 'h', 'pc-ov1.5'),
                "Over 1.5": get_statistic("over", 'h', 'pc-ov1.5'),
                "Under 2.5": get_statistic("under", 'h', 'pc-ov2.5'),
                "Over 2.5": get_statistic("over", 'h', 'pc-ov2.5'),
                "Under 3.5": get_statistic("under", 'h', 'pc-ov3.5'),
                "Over 3.5": get_statistic("over", 'h', 'pc-ov3.5'),
                "Yes": get_statistic("under", 'h', 'bottom_chart'),
                "No": get_statistic("over", 'h', 'bottom_chart'),
            }

            away_stats = {
                "Under 1.5": get_statistic("under", 'a', 'pc-ov1.5'),
                "Over 1.5": get_statistic("over", 'a', 'pc-ov1.5'),
                "Under 2.5": get_statistic("under", 'a', 'pc-ov2.5'),
                "Over 2.5": get_statistic("over", 'a', 'pc-ov2.5'),
                "Under 3.5": get_statistic("under", 'a', 'pc-ov3.5'),
                "Over 3.5": get_statistic("over", 'a', 'pc-ov3.5'),
                "Yes": get_statistic("under", 'a', 'bottom_chart'),
                "No": get_statistic("over", 'a', 'bottom_chart'),
            }

            matches.append({
                "Match Link": match_url,
                "Home Team Name": home_team_name,
                "Away Team Name": away_team_name,
                "Home Matches Played": home_matches_played,
                "Home Standing": home_standing,
                "Away Matches Played": away_matches_played,
                "Away Standing": away_standing,
                "Match Time": match_time,
                **home_stats,
                **away_stats
            })
        except Exception as e:
            logging.error(f"Error parsing {match_url}: {e}")

    return matches

def scrape_all_pages(start_url):
    all_matches = []
    next_page_url = start_url

    driver = setup_selenium()

    try:
        while next_page_url:
            html_content = get_page_content(next_page_url, driver)
            if not html_content:
                break

            soup = BeautifulSoup(html_content, 'html.parser')
            matches = parse_match_details(soup, driver)
            all_matches.extend(matches)

            load_more_button = soup.select_one('#mrows span')
            if load_more_button:
                next_page_url = load_more_button.get('data-next-page-url')
                if next_page_url:
                    if not next_page_url.startswith('http'):
                        next_page_url = f"https://www.example.com{next_page_url}"
                    time.sleep(page_load_delay)
                else:
                    next_page_url = None
            else:
                next_page_url = None
    finally:
        driver.quit()

    return all_matches

def main():
    matches = scrape_all_pages(base_url)
    df = pd.DataFrame(matches)
    df.to_csv('football_matches.csv', index=False)
    print("Data saved to football_matches.csv")

if __name__ == "__main__":
    main()
原因分析
  1. 分页/加载逻辑错误:当前代码尝试通过读取#mrows span的data-next-page-url获取下一页链接,但该元素可能定位错误,或者网站采用滚动加载而非点击加载更多的机制,导致仅获取了初始加载的7场比赛。
  2. 页面加载不充分:固定的time.sleep()延迟不足以让页面完全加载后续内容,导致DOM中未渲染出更多比赛的元素。
  3. 反爬机制限制:网站可能检测到爬虫行为,限制了返回的数据量,或者对请求频率、自动化标识进行了拦截。
  4. 选择器范围有限:.contentmiddle a[itemprop="url"]选择器可能仅匹配了初始加载的比赛链接,后续加载的比赛元素不在该容器或属性范围内。
解决方案

1. 修复动态加载逻辑

若网站为滚动加载:

修改get_page_content函数,模拟滚动到底部直到无新内容加载:

def get_page_content(url, driver, page_load_delay=2):
    driver.get(url)
    # 滚动加载全部内容
    last_height = driver.execute_script("return document.body.scrollHeight")
    while True:
        driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
        time.sleep(page_load_delay)
        new_height = driver.execute_script("return document.body.scrollHeight")
        if new_height == last_height:
            break
        last_height = new_height
    return driver.page_source

若网站为点击加载更多:

替换通过属性取URL的方式,直接用Selenium点击按钮加载内容:

from selenium.common.exceptions import NoSuchElementException
from selenium.webdriver.common.by import By

def scrape_all_pages(start_url):
    all_matches = []
    driver = setup_selenium()

    try:
        driver.get(start_url)
        while True:
            # 滚动到页面底部确保加载更多按钮可见
            driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
            time.sleep(2)
            # 获取当前页面的比赛数据
            html_content = driver.page_source
            soup = BeautifulSoup(html_content, 'html.parser')
            matches = parse_match_details(soup, driver)
            all_matches.extend(matches)
            # 尝试点击加载更多按钮
            try:
                load_more_btn = driver.find_element(By.CSS_SELECTOR, '#mrows span')
                driver.execute_script("arguments[0].scrollIntoView();", load_more_btn)
                time.sleep(1)
                load_more_btn.click()
                time.sleep(2)
                # 检查按钮是否还存在,不存在则停止
                if not driver.find_elements(By.CSS_SELECTOR, '#mrows span'):
                    break
            except NoSuchElementException:
                break
    finally:
        driver.quit()

    return all_matches

2. 替换固定延迟为显式等待

使用WebDriverWait等待关键元素加载完成,确保页面内容完全渲染:

from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.by import By

def get_page_content(url, driver):
    driver.get(url)
    # 等待比赛列表加载完成
    WebDriverWait(driver, 10).until(
        EC.presence_of_all_elements_located((By.CSS_SELECTOR, '.contentmiddle a[itemprop="url"]'))
    )
    # 滚动加载全部内容
    last_height = driver.execute_script("return document.body.scrollHeight")
    while True:
        driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
        time.sleep(2)
        new_height = driver.execute_script("return document.body.scrollHeight")
        if new_height == last_height:
            break
        last_height = new_height
    # 再次等待新内容加载完成
    WebDriverWait(driver, 10).until(
        EC.presence_of_all_elements_located((By.CSS_SELECTOR, '.contentmiddle a[itemprop="url"]'))
    )
    return driver.page_source

3. 优化反爬规避策略

  • 增强Selenium的反检测能力:
def setup_selenium():
    options = Options()
    options.add_argument('--headless')
    options.add_argument('--no-sandbox')
    options.add_argument('--disable-dev-shm-usage')
    options.add_argument("--window-size=1920,1080")
    options.add_argument("--disable-infobars")
    options.add_argument("--disable-popup-blocking")
    options.add_argument("--ignore-certificate-errors")
    options.add_argument("--incognito")
    options.add_argument('--disable-blink-features=AutomationControlled')
    options.add_experimental_option("excludeSwitches", ["enable-automation"])
    options.add_experimental_option('useAutomationExtension', False)
    options.add_argument(f'--user-agent={choice(user_agents)}')
    driver = gs.Chrome(options=options)
    # 隐藏自动化标识
    driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})")
    return driver
  • 增加请求间隔至3-5秒,避免请求过于频繁;
  • 扩充User-Agent列表,增加更多不同浏览器的标识。

4. 验证并修正选择器

使用浏览器开发者工具检查页面DOM结构,确认.contentmiddle a[itemprop="url"]是否能匹配所有比赛链接。若后续加载的比赛元素不在该容器内,需更新选择器为正确的定位规则。


内容的提问来源于stack exchange,提问作者Akingeneye Benjamin

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 17:22:02