You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Selenium和Python爬虫时无法点击「Next」按钮的问题求助

无法点击分页「Next」按钮的爬虫问题解决

我需要爬取https://skills.education.nsw.gov.au/nsw-fee-free的课程及其提供商信息,目标是遍历每门课程并访问提供商列表;若提供商列表有多页,需点击「Next」按钮获取全部数据,但目前无法成功点击该按钮。

以下是我使用的代码:

import requests
from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import NoSuchElementException
import csv
import time

def extract_provider_data(driver):
    provider_names = []

    while True:
        try:
            providers = WebDriverWait(driver, 10).until(
                EC.presence_of_all_elements_located((By.CSS_SELECTOR, "div.provider-name a"))
            )
            for provider in providers:
                provider_names.append(provider.text)
            
            # Ensure the next page button is clickable before attempting to click
            next_page = WebDriverWait(driver, 5).until(
                EC.element_to_be_clickable((By.CSS_SELECTOR, 'li.paginationjs-page.J-paginationjs-page:not(.paginationjs-current) a'))
            )
            if next_page:
                next_page.click()
                time.sleep(3)
            else:
                break
        except NoSuchElementException:
            break
        except Exception as e:
            print(f"Error while clicking the next page button: {str(e)}")
            break

    return provider_names

def extract_data(driver):
    soup = BeautifulSoup(driver.page_source, 'html.parser')
    data = []

    course_name = soup.find('div', class_='course-name')
    course_type = soup.find('div', class_='course-type', id='course-label-default')
    delivery_mode = soup.find('div', {'class': 'col data', 'data-th': 'delivery-mode'})
    industry_name = soup.find('div', class_='category-heading container industry-info').find('h2').text.strip()
    
    provider_link = WebDriverWait(driver, 10).until(
        EC.presence_of_element_located((By.CSS_SELECTOR, 'div.provider-location a.provider-location-link'))
    )
    provider_link.click()

    provider_names = extract_provider_data(driver)
    driver.back()
    time.sleep(3)  # Allow time for the page to reload

    for provider in provider_names:
        row = {
            'Industry': industry_name,
            'Course Name': course_name.text.strip(),
            'Course Type': course_type.text.strip(),
            'Delivery Mode': delivery_mode.text.strip(),
            'Provider Name': provider
        }
        data.append(row)

    return data

# Extract list of URLs using requests and BeautifulSoup
URL = 'https://skills.education.nsw.gov.au/nsw-fee-free'
response = requests.get(URL)
soup = BeautifulSoup(response.content, 'html.parser')

a_tags = soup.find_all('a', class_='industry')
industry_urls = [tag['href'] for tag in a_tags]

# Set up the webdriver
driver = webdriver.Chrome()  # Ensure chromedriver is in your PATH or provide the path explicitly

# List to store all scraped data
all_data = []

# Counter for courses
course_counter = 0

# Visit each URL and extract data
for industry_url in industry_urls:
    driver.get(industry_url)
    time.sleep(3)  # Allow time for the page to load
    page_data = extract_data(driver)
    all_data.extend(page_data)
    course_counter += 1

    if course_counter >= 10:  # Limit to the first 10 courses
        break

# Close the browser when done
driver.quit()

# Save the data to CSV
with open('output.csv', 'w', newline='', encoding='utf-8') as csvfile:
    fieldnames = ['Industry', 'Course Name', 'Course Type', 'Delivery Mode', 'Provider Name']
    writer = csv.DictWriter(csvfile, fieldnames=fieldnames)

    writer.writeheader()
    for row in all_data:
        writer.writerow(row)

print('Data has been successfully saved to output.csv')

问题分析

无法点击「Next」按钮通常由以下几个原因导致:

  • CSS选择器不准确:原选择器可能匹配到多个元素或未正确定位到「Next」按钮
  • 元素未完全加载:AJAX动态加载分页按钮时,显式等待的条件或时长不足
  • 元素不可见/被遮挡:分页按钮可能在视窗之外,需要滚动到可见区域才能点击
  • 异常捕获逻辑缺陷:NoSuchElementException未覆盖所有终止循环的场景

解决方案

1. 修正「Next」按钮的CSS选择器

观察页面结构,「Next」按钮带有明确的类标识,可使用更精准的选择器:

li.paginationjs-next a

2. 确保元素可见再点击

使用EC.visibility_of_element_located替代element_to_be_clickable,并添加滚动操作保证按钮在视窗内:

next_page = WebDriverWait(driver, 10).until(
    EC.visibility_of_element_located((By.CSS_SELECTOR, 'li.paginationjs-next a'))
)
# 滚动到按钮可见区域
driver.execute_script("arguments[0].scrollIntoView(true);", next_page)
next_page.click()

3. 优化循环终止逻辑

检查按钮是否处于禁用状态,避免无效点击:

if 'disabled' in next_page.parent.get_attribute('class'):
    break

4. 替换time.sleep为显式等待

使用显式等待替代固定时长睡眠,提升脚本稳定性:

# 等待页面刷新完成
WebDriverWait(driver, 10).until(
    EC.staleness_of(providers[0])  # 等待上一页元素失效
)

修改后的完整代码

import requests
from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import NoSuchElementException, TimeoutException
import csv

def extract_provider_data(driver):
    provider_names = []

    while True:
        try:
            # 等待提供商列表加载完成
            providers = WebDriverWait(driver, 10).until(
                EC.presence_of_all_elements_located((By.CSS_SELECTOR, "div.provider-name a"))
            )
            for provider in providers:
                provider_names.append(provider.text.strip())
            
            # 定位Next按钮
            try:
                next_page = WebDriverWait(driver, 5).until(
                    EC.visibility_of_element_located((By.CSS_SELECTOR, 'li.paginationjs-next a'))
                )
                # 滚动到按钮可见
                driver.execute_script("arguments[0].scrollIntoView(true);", next_page)
                # 检查是否为最后一页(按钮是否禁用)
                if 'disabled' in next_page.parent.get_attribute('class'):
                    break
                next_page.click()
                # 等待页面刷新
                WebDriverWait(driver, 10).until(
                    EC.staleness_of(providers[0])
                )
            except TimeoutException:
                # 没有Next按钮,终止循环
                break
        except Exception as e:
            print(f"处理提供商分页时出错: {str(e)}")
            break

    return provider_names

def extract_data(driver):
    # 等待课程页面加载完成
    WebDriverWait(driver, 10).until(
        EC.presence_of_element_located((By.CSS_SELECTOR, 'div.course-name'))
    )
    soup = BeautifulSoup(driver.page_source, 'html.parser')
    data = []

    course_name = soup.find('div', class_='course-name').text.strip()
    course_type = soup.find('div', class_='course-type', id='course-label-default').text.strip()
    delivery_mode = soup.find('div', {'class': 'col data', 'data-th': 'delivery-mode'}).text.strip()
    industry_name = soup.find('div', class_='category-heading container industry-info').find('h2').text.strip()
    
    # 点击提供商链接
    provider_link = WebDriverWait(driver, 10).until(
        EC.element_to_be_clickable((By.CSS_SELECTOR, 'div.provider-location a.provider-location-link'))
    )
    provider_link.click()

    provider_names = extract_provider_data(driver)
    driver.back()
    # 等待课程页面重新加载
    WebDriverWait(driver, 10).until(
        EC.presence_of_element_located((By.CSS_SELECTOR, 'div.course-name'))
    )

    for provider in provider_names:
        row = {
            'Industry': industry_name,
            'Course Name': course_name,
            'Course Type': course_type,
            'Delivery Mode': delivery_mode,
            'Provider Name': provider
        }
        data.append(row)

    return data

# 获取行业课程URL列表
URL = 'https://skills.education.nsw.gov.au/nsw-fee-free'
response = requests.get(URL)
soup = BeautifulSoup(response.content, 'html.parser')
industry_urls = [tag['href'] for tag in soup.find_all('a', class_='industry')]

# 初始化浏览器
driver = webdriver.Chrome()
all_data = []
course_counter = 0

for industry_url in industry_urls:
    driver.get(industry_url)
    page_data = extract_data(driver)
    all_data.extend(page_data)
    course_counter += 1
    if course_counter >= 10:
        break

driver.quit()

# 保存数据到CSV
with open('output.csv', 'w', newline='', encoding='utf-8') as csvfile:
    fieldnames = ['Industry', 'Course Name', 'Course Type', 'Delivery Mode', 'Provider Name']
    writer = csv.DictWriter(csvfile, fieldnames=fieldnames)
    writer.writeheader()
    for row in all_data:
        writer.writerow(row)

print('数据已成功保存到output.csv')

内容的提问来源于stack exchange,提问作者Chuixi

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.07 13:44:51