提取RFI站点分页HTML表格问题求助(Selenium代码调试)
解决RFI站点分页爬取问题
你遇到的核心问题是下一页按钮定位不准确,且未判断按钮的可用状态,导致无法触发分页跳转。以下是修改后的可运行代码及关键调整说明:
修改后的完整代码
from selenium import webdriver from selenium.webdriver.edge.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pandas as pd from bs4 import BeautifulSoup # 初始化WebDriver service = Service('D:/Downloads/edgedriver_win64/msedgedriver.exe') driver = webdriver.Edge(service=service) driver.get('https://www.rfi.it/it/stazioni.html') # 处理Cookie弹窗 try: accept_cookies = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, '//button[text()="Accetta tutti i cookie"]')) ) accept_cookies.click() except Exception as e: print(f"Cookie按钮处理失败: {e}") # 切换到LISTA标签页 lista_tab = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, '//a[text()="LISTA"]')) ) driver.execute_script("arguments[0].click();", lista_tab) data = [] current_page = 1 while True: print(f"正在处理第 {current_page} 页") # 等待表格加载完成后再解析 WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.CLASS_NAME, 'table-striped')) ) # 解析页面表格 soup = BeautifulSoup(driver.page_source, 'html.parser') table = soup.find('table', class_='table table-striped table-hover') if not table: print(f"第 {current_page} 页未找到表格,停止爬取") break # 提取表头(仅第一页) if current_page == 1: headers = [th.text.strip() for th in table.find_all('th')] data.append(headers) # 提取表格行数据 for row in table.find_all('tr'): cols = row.find_all('td') if cols: row_data = [col.text.strip() for col in cols] data.append(row_data) # 定位并点击下一页按钮 try: # 仅定位可用的下一页按钮(排除禁用状态) next_button = WebDriverWait(driver, 10).until( EC.and_( EC.element_to_be_clickable((By.XPATH, '//button[contains(@class, "pagination--next-btn") and not(@disabled)]')), EC.visibility_of_element_located((By.XPATH, '//button[contains(@class, "pagination--next-btn") and not(@disabled)]')) ) ) driver.execute_script("arguments[0].click();", next_button) # 等待页码更新,确保页面跳转完成 WebDriverWait(driver, 10).until( EC.text_to_be_present_in_element((By.CLASS_NAME, 'pagination--current'), str(current_page + 1)) ) current_page += 1 except Exception as e: print(f"无更多页面或下一页按钮不可用,停止爬取。错误信息: {e}") break # 关闭浏览器 driver.quit() # 生成DataFrame并展示结果 if data: df = pd.DataFrame(data[1:], columns=data[0]) df = df.dropna(how='all') print(df) else: print("未收集到任何数据")
关键修改说明
- 修正下一页按钮定位:添加
not(@disabled)条件,确保只定位处于可用状态的下一页按钮,避免点击已禁用的无效按钮。 - 优化等待逻辑:
- 等待表格加载完成后再解析页面,避免因页面未渲染完全导致的空表格问题。
- 等待页码元素更新完成后再进入下一轮循环,确保分页跳转已成功完成。
- 简化元素匹配:将模糊的文本匹配改为精准的
text()匹配,减少定位到相似元素的概率。
内容的提问来源于stack exchange,提问作者Tan Phan
相关产品推荐
相关产品推荐

