You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Python+Selenium爬取无URL变化无下一页按钮的多页网站

问题描述

我需要用Python和Selenium爬取一个切换页面时URL不变化、也没有NEXT按钮的网站,总共要提取63846页的数据。我试过用Selenium的Select组件但没成功,现在的代码只能获取第一页的数据。网站采用paginationjs分页结构,点击页码对应的li元素就能切换页面。以下是我的现有代码和分页HTML结构:

现有Python代码:

from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.support.ui import Select
import pandas as pd
import re
import math
from time import sleep
from selenium import webdriver
from selenium.webdriver.common.by import By
from bs4 import BeautifulSoup
from selenium.webdriver.support import expected_conditions as EC

#SET UP DRIVER
driver = webdriver.Firefox()
extension_path = "/home/work04/.mozilla/firefox/vyrx65nr.default-release/extensions/{e58d3966-3d76-4cd9-8552-1582fbc800c1}.xpi"
driver.install_addon(extension_path, temporary=True)
driver.maximize_window()

driver.get("URL WEBSITE")
sleep(4)
                
#CLICKS
driver.find_element(By.XPATH, '//*[@id="page"]/div[4]/div[2]/button').click()    #Cookie
sleep(2)  
driver.find_element(By.NAME, 'tipoSituacao').click()
sleep(1)
driver.find_element(By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/article/div[2]/div/div/form/div/div[2]/div[3]/div/select/option[2]').click()
sleep(1)
driver.find_element(By.NAME, 'situacao').click()
sleep(1)
driver.find_element(By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/article/div[2]/div/div/form/div/div[2]/div[4]/div/select/option[1]').click()
sleep(1)
driver.find_element(By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/article/div[2]/div/div/form/div/div[4]/div[2]/button').click()
sleep(2)  

#WAIT 
element = WebDriverWait(driver, 120).until(EC.presence_of_element_located((By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/div[2]/div[1]/div/div/div[2]/div')))
assert element.is_displayed()

#OBJECT HTML
page = driver.page_source
soup = BeautifulSoup(page, 'html.parser')
sleep(10)


data_list = []

#Data in DR
dr = soup.find_all(class_=re.compile('CLASS NAME I WANT'))

#DATA
for x in range(len(dr)):
    data = {}

    data['xxx'] = x.find('h4').get_text()

    col_md_4_divs = x.find_all(class_='col-md-4')
    data['xxx'] = col_md_4_divs[0].b.next_sibling.strip()
    # data['xx'] = col_md_4_divs[0].find(string=re.compile('([A-Z]{2})')).get_text().strip()
    data['xxxxxx'] = col_md_4_divs[3].b.next_sibling.strip()
    data['xxxxxx'] = col_md_4_divs[3].find_next(class_='col-md').b.next_sibling.strip()

    especialidades_div = str(x.find(class_='col-md-12', style='display: flex;'))
    print(xxxxxxxxxx)
    if especialidades_div.find('</span>') != -1:
        especialidades = str(especialidades_div)
        print(f"esp {especialidades}")
        data['especialidades'] = especialidades

    endereco_div = x.find_all(class_='col-md-7')
    data['endereco'] = endereco_div[0].b.next_sibling.strip()

    telefone_div = x.find_all(class_='row telefone')
    data['telefone'] = telefone_div[0].b.next_sibling.strip()

    data_list.append(data)

#CSV
df = pd.DataFrame(data_list)
csv_file_path = '------'
df.to_csv(csv_file_path)
print(df)


driver.quit()

分页HTML结构:

<div class="paginationjs-pages">
 <ul>
  <li class="paginationjs-page J-paginationjs-page active" data-num="1">
   <a>
    1
   </a>
  </li>
  <li class="paginationjs-page J-paginationjs-page" data-num="2">
   <a href="">
    2
   </a>
  </li>
  <li class="paginationjs-page J-paginationjs-page" data-num="3">
   <a href="">
    3
   </a>
  </li>
  <li class="paginationjs-page J-paginationjs-page" data-num="4">
   <a href="">
    4
   </a>
  </li>
  <li class="paginationjs-page J-paginationjs-page" data-num="5">
   <a href="">
    5
   </a>
  </li>
  <li class="paginationjs-ellipsis disabled">
   <a>
    ...
   </a>
  </li>
  <li class="paginationjs-page paginationjs-last J-paginationjs-page" data-num="63846">
   <a href="">
    63846
   </a>
  </li>
 </ul>
</div>

解决方案

针对paginationjs分页结构,我们可以通过定位页码元素并点击实现翻页,同时优化等待逻辑和异常处理,确保批量爬取的稳定性。

1. 修改后的完整代码

from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.support.ui import Select
import pandas as pd
import re
import random
from time import sleep
from selenium import webdriver
from selenium.webdriver.common.by import By
from bs4 import BeautifulSoup
from selenium.webdriver.support import expected_conditions as EC

def extract_page_data(driver):
    """提取当前页面数据的封装函数"""
    page = driver.page_source
    soup = BeautifulSoup(page, 'html.parser')
    data_list = []
    
    dr = soup.find_all(class_=re.compile('CLASS NAME I WANT'))
    for item in dr:
        data = {}
        try:
            # 提取标题
            h4_tag = item.find('h4')
            if h4_tag:
                data['xxx'] = h4_tag.get_text(strip=True)
            
            # 提取col-md-4区域数据
            col_md_4_divs = item.find_all(class_='col-md-4')
            if col_md_4_divs:
                data['xxx'] = col_md_4_divs[0].b.next_sibling.strip()
                if len(col_md_4_divs) >= 4:
                    data['xxxxxx'] = col_md_4_divs[3].b.next_sibling.strip()
                    next_col_md = col_md_4_divs[3].find_next(class_='col-md')
                    if next_col_md:
                        data['xxxxxx'] = next_col_md.b.next_sibling.strip()
            
            # 提取专业信息
            especialidades_div = item.find(class_='col-md-12', style='display: flex;')
            if especialidades_div and '</span>' in str(especialidades_div):
                data['especialidades'] = str(especialidades_div)
            
            # 提取地址和电话
            endereco_div = item.find_all(class_='col-md-7')
            if endereco_div:
                data['endereco'] = endereco_div[0].b.next_sibling.strip()
            
            telefone_div = item.find_all(class_='row telefone')
            if telefone_div:
                data['telefone'] = telefone_div[0].b.next_sibling.strip()
            
            data_list.append(data)
        except Exception as e:
            print(f"提取单条数据失败: {e}")
            continue
    return data_list

# 初始化驱动
driver = webdriver.Firefox()
extension_path = "/home/work04/.mozilla/firefox/vyrx65nr.default-release/extensions/{e58d3966-3d76-4cd9-8552-1582fbc800c1}.xpi"
driver.install_addon(extension_path, temporary=True)
driver.maximize_window()

driver.get("URL WEBSITE")

# 处理Cookie和筛选条件(用显式等待替代固定sleep)
WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.XPATH, '//*[@id="page"]/div[4]/div[2]/button'))).click()

WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.NAME, 'tipoSituacao'))).click()
WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/article/div[2]/div/div/form/div/div[2]/div[3]/div/select/option[2]'))).click()

WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.NAME, 'situacao'))).click()
WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/article/div[2]/div/div/form/div/div[2]/div[4]/div/select/option[1]'))).click()

WebDriverWait(driver, 20).until(EC.element_to_be_clickable((By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/article/div[2]/div/div/form/div/div[4]/div[2]/button'))).click()

# 等待第一页加载完成
WebDriverWait(driver, 120).until(EC.presence_of_element_located((By.CLASS_NAME, 'paginationjs-pages')))

all_data = []
total_pages = 63846
# 可添加断点续爬逻辑:读取已爬取页码,从该页码开始
start_page = 1

for page_num in range(start_page, total_pages + 1):
    print(f"正在爬取第 {page_num} 页")
    # 提取当前页数据
    page_data = extract_page_data(driver)
    all_data.extend(page_data)
    
    # 最后一页无需翻页
    if page_num == total_pages:
        break
    
    try:
        # 定位下一页元素(通过data-num属性)
        next_page_selector = f'li.paginationjs-page[data-num="{page_num + 1}"]'
        next_page_li = WebDriverWait(driver, 20).until(
            EC.element_to_be_clickable((By.CSS_SELECTOR, next_page_selector))
        )
        # 滚动到元素位置,避免视窗外无法点击
        driver.execute_script("arguments[0].scrollIntoView(true);", next_page_li)
        next_page_li.click()
        
        # 验证页面切换成功:等待新页码变为活跃状态
        WebDriverWait(driver, 20).until(
            EC.presence_of_element_located((By.CSS_SELECTOR, f'li.paginationjs-page.active[data-num="{page_num + 1}"]'))
        )
        # 等待数据区域刷新
        WebDriverWait(driver, 20).until(
            EC.presence_of_element_located((By.XPATH, '/html/body/div[1]/div[1]/section[2]/div/div/div/div[2]/div[1]/div/div/div[2]/div'))
        )
        # 随机等待,降低反爬风险
        sleep(random.uniform(1, 3))
    except Exception as e:
        print(f"翻页到第 {page_num + 1} 页失败: {e}")
        # 保存已爬取数据后退出
        df = pd.DataFrame(all_data)
        df.to_csv(f'临时爬取结果_{page_num}页.csv', index=False, encoding='utf-8-sig')
        print(f"已保存临时数据到 临时爬取结果_{page_num}页.csv")
        break

# 保存全部数据
df = pd.DataFrame(all_data)
csv_file_path = '完整爬取结果.csv'
df.to_csv(csv_file_path, index=False, encoding='utf-8-sig')
print(f"爬取完成,共 {len(all_data)} 条数据,已保存到 {csv_file_path}")

driver.quit()

2. 关键优化说明

  • 显式等待替代固定sleep:避免不必要的等待,同时确保元素加载完成后再操作,提升稳定性
  • 数据提取封装:把页面数据提取逻辑单独封装,代码更清晰,便于维护
  • 异常捕获:单个数据或页面出错时不中断整个爬取流程,同时保存临时数据
  • 滚动定位:通过JavaScript滚动到页码元素,解决视窗外元素无法点击的问题
  • 页码验证:翻页后等待活跃页码更新,确保页面确实切换成功
  • 反爬优化:加入随机等待时间,降低被网站检测到的风险

3. 额外建议

  • 爬取6万+页面耗时极长,建议添加断点续爬逻辑:每次保存数据时记录已爬取页码,下次启动从该页码继续
  • 可以使用多线程/多进程提升爬取速度,但需注意控制请求频率,避免触发反爬
  • 若遇到页码被省略号隐藏的情况,可以尝试直接调用paginationjs的JS方法翻页,比如:driver.execute_script("window.paginationjsInstance.goToPage({page_num})")(需查看网站全局的paginationjs实例名称)

内容的提问来源于stack exchange,提问作者Julia

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.12 21:15:01