Selenium脚本仅爬取Forebet7场比赛数据的原因与解决方案
问题描述
我正在使用Selenium开展网页爬取项目,从足球预测网站(以EXAMPLE指代FOREBET)爬取足球比赛数据。但网页上明明列出了更多比赛,我的脚本却仅能获取7场比赛的数据。相关代码如下:
import time from bs4 import BeautifulSoup import pandas as pd import logging import google_colab_selenium as gs from selenium.webdriver.chrome.options import Options from random import choice # Configurations request_interval = 2 # seconds page_load_delay = 2 # seconds base_url = "https://www.example.com/en/football-tips-and-predictions-for-today/predictions-1x2" # User agents list (non-mobile) user_agents = [ 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.3', 'Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/56.0.2924.87 Safari/537.36', 'Mozilla/5.0 (Windows NT 6.1; WOW64; Trident/7.0; AS; rv:11.0) like Gecko', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_6) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/14.0.1 Safari/605.1.15', 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/88.0.4324.96 Safari/537.36' ] # Configure logging logging.basicConfig(level=logging.INFO) # Setup Chrome options and driver def setup_selenium(): options = Options() options.add_argument('--headless') options.add_argument('--no-sandbox') options.add_argument('--disable-dev-shm-usage') options.add_argument("--window-size=1920,1080") options.add_argument("--disable-infobars") options.add_argument("--disable-popup-blocking") options.add_argument("--ignore-certificate-errors") options.add_argument("--incognito") options.add_argument(f'--user-agent={choice(user_agents)}') driver = gs.Chrome(options=options) return driver def get_page_content(url, driver, request_interval=2, page_load_delay=2): driver.get(url) time.sleep(request_interval) html_content = driver.page_source time.sleep(page_load_delay) return html_content def parse_match_details(soup, driver): matches = [] match_links = [a['href'] for a in soup.select('.contentmiddle a[itemprop="url"]')] for link in match_links: match_url = f"https://www.example.com{link}" html_content = get_page_content(match_url, driver) match_soup = BeautifulSoup(html_content, 'lxml') if not match_soup: continue try: home_team_name = match_soup.select_one("[itemprop='homeTeam'] [itemprop='url'] span").text.strip() away_team_name = match_soup.select_one("[itemprop='awayTeam'] [itemprop='url'] span").text.strip() home_matches_played = match_soup.select_one(".os_played_games_main span[data-team='h']").text.strip() away_matches_played = match_soup.select_one(".os_played_games_main span[data-team='a']").text.strip() home_standing = match_soup.select_one("span.teamtableleft").text.strip() away_standing = match_soup.select_one("span.teamtableright").text.strip() match_time = match_soup.select_one("[itemprop='startDate'] div").text.strip() recent_matches = match_soup.select_one("div.os_goals_section3_container") home_and_away_goals = recent_matches def get_statistic(selector, data_team, data_stat): stat = home_and_away_goals.select_one(f"[data-team='{data_team}'][data-stat='{data_stat}'] span.__{selector}") return stat.text.strip() if stat else "" home_stats = { "Under 1.5": get_statistic("under", 'h', 'pc-ov1.5'), "Over 1.5": get_statistic("over", 'h', 'pc-ov1.5'), "Under 2.5": get_statistic("under", 'h', 'pc-ov2.5'), "Over 2.5": get_statistic("over", 'h', 'pc-ov2.5'), "Under 3.5": get_statistic("under", 'h', 'pc-ov3.5'), "Over 3.5": get_statistic("over", 'h', 'pc-ov3.5'), "Yes": get_statistic("under", 'h', 'bottom_chart'), "No": get_statistic("over", 'h', 'bottom_chart'), } away_stats = { "Under 1.5": get_statistic("under", 'a', 'pc-ov1.5'), "Over 1.5": get_statistic("over", 'a', 'pc-ov1.5'), "Under 2.5": get_statistic("under", 'a', 'pc-ov2.5'), "Over 2.5": get_statistic("over", 'a', 'pc-ov2.5'), "Under 3.5": get_statistic("under", 'a', 'pc-ov3.5'), "Over 3.5": get_statistic("over", 'a', 'pc-ov3.5'), "Yes": get_statistic("under", 'a', 'bottom_chart'), "No": get_statistic("over", 'a', 'bottom_chart'), } matches.append({ "Match Link": match_url, "Home Team Name": home_team_name, "Away Team Name": away_team_name, "Home Matches Played": home_matches_played, "Home Standing": home_standing, "Away Matches Played": away_matches_played, "Away Standing": away_standing, "Match Time": match_time, **home_stats, **away_stats }) except Exception as e: logging.error(f"Error parsing {match_url}: {e}") return matches def scrape_all_pages(start_url): all_matches = [] next_page_url = start_url driver = setup_selenium() try: while next_page_url: html_content = get_page_content(next_page_url, driver) if not html_content: break soup = BeautifulSoup(html_content, 'html.parser') matches = parse_match_details(soup, driver) all_matches.extend(matches) load_more_button = soup.select_one('#mrows span') if load_more_button: next_page_url = load_more_button.get('data-next-page-url') if next_page_url: if not next_page_url.startswith('http'): next_page_url = f"https://www.example.com{next_page_url}" time.sleep(page_load_delay) else: next_page_url = None else: next_page_url = None finally: driver.quit() return all_matches def main(): matches = scrape_all_pages(base_url) df = pd.DataFrame(matches) df.to_csv('football_matches.csv', index=False) print("Data saved to football_matches.csv") if __name__ == "__main__": main()
原因分析
- 分页/加载逻辑错误:当前代码尝试通过读取
#mrows span的data-next-page-url获取下一页链接,但该元素可能定位错误,或者网站采用滚动加载而非点击加载更多的机制,导致仅获取了初始加载的7场比赛。 - 页面加载不充分:固定的
time.sleep()延迟不足以让页面完全加载后续内容,导致DOM中未渲染出更多比赛的元素。 - 反爬机制限制:网站可能检测到爬虫行为,限制了返回的数据量,或者对请求频率、自动化标识进行了拦截。
- 选择器范围有限:
.contentmiddle a[itemprop="url"]选择器可能仅匹配了初始加载的比赛链接,后续加载的比赛元素不在该容器或属性范围内。
解决方案
1. 修复动态加载逻辑
若网站为滚动加载:
修改get_page_content函数,模拟滚动到底部直到无新内容加载:
def get_page_content(url, driver, page_load_delay=2): driver.get(url) # 滚动加载全部内容 last_height = driver.execute_script("return document.body.scrollHeight") while True: driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(page_load_delay) new_height = driver.execute_script("return document.body.scrollHeight") if new_height == last_height: break last_height = new_height return driver.page_source
若网站为点击加载更多:
替换通过属性取URL的方式,直接用Selenium点击按钮加载内容:
from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.common.by import By def scrape_all_pages(start_url): all_matches = [] driver = setup_selenium() try: driver.get(start_url) while True: # 滚动到页面底部确保加载更多按钮可见 driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(2) # 获取当前页面的比赛数据 html_content = driver.page_source soup = BeautifulSoup(html_content, 'html.parser') matches = parse_match_details(soup, driver) all_matches.extend(matches) # 尝试点击加载更多按钮 try: load_more_btn = driver.find_element(By.CSS_SELECTOR, '#mrows span') driver.execute_script("arguments[0].scrollIntoView();", load_more_btn) time.sleep(1) load_more_btn.click() time.sleep(2) # 检查按钮是否还存在,不存在则停止 if not driver.find_elements(By.CSS_SELECTOR, '#mrows span'): break except NoSuchElementException: break finally: driver.quit() return all_matches
2. 替换固定延迟为显式等待
使用WebDriverWait等待关键元素加载完成,确保页面内容完全渲染:
from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.common.by import By def get_page_content(url, driver): driver.get(url) # 等待比赛列表加载完成 WebDriverWait(driver, 10).until( EC.presence_of_all_elements_located((By.CSS_SELECTOR, '.contentmiddle a[itemprop="url"]')) ) # 滚动加载全部内容 last_height = driver.execute_script("return document.body.scrollHeight") while True: driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(2) new_height = driver.execute_script("return document.body.scrollHeight") if new_height == last_height: break last_height = new_height # 再次等待新内容加载完成 WebDriverWait(driver, 10).until( EC.presence_of_all_elements_located((By.CSS_SELECTOR, '.contentmiddle a[itemprop="url"]')) ) return driver.page_source
3. 优化反爬规避策略
- 增强Selenium的反检测能力:
def setup_selenium(): options = Options() options.add_argument('--headless') options.add_argument('--no-sandbox') options.add_argument('--disable-dev-shm-usage') options.add_argument("--window-size=1920,1080") options.add_argument("--disable-infobars") options.add_argument("--disable-popup-blocking") options.add_argument("--ignore-certificate-errors") options.add_argument("--incognito") options.add_argument('--disable-blink-features=AutomationControlled') options.add_experimental_option("excludeSwitches", ["enable-automation"]) options.add_experimental_option('useAutomationExtension', False) options.add_argument(f'--user-agent={choice(user_agents)}') driver = gs.Chrome(options=options) # 隐藏自动化标识 driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})") return driver
- 增加请求间隔至3-5秒,避免请求过于频繁;
- 扩充User-Agent列表,增加更多不同浏览器的标识。
4. 验证并修正选择器
使用浏览器开发者工具检查页面DOM结构,确认.contentmiddle a[itemprop="url"]是否能匹配所有比赛链接。若后续加载的比赛元素不在该容器内,需更新选择器为正确的定位规则。
内容的提问来源于stack exchange,提问作者Akingeneye Benjamin
相关产品推荐
相关产品推荐

