Selenium爬取clutch.co时分页功能失效问题求助
爬取Clutch.co数据时无法点击「下一页」翻页的问题解决
问题描述
爬取Clutch.co美国网页开发者列表(https://clutch.co/us/web-developers)时,脚本仅能获取第一页数据,之后浏览器自动关闭,无法触发「下一页」按钮完成翻页。已尝试添加等待机制但无效,运行时可见浏览器滚动至页面底部后直接关闭。
问题代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pandas as pd import time website = "https://clutch.co/us/web-developers" options = webdriver.ChromeOptions() options.add_experimental_option("detach", False) driver = webdriver.Chrome(options=options) driver.get(website) wait = WebDriverWait(driver, 10) company_elements = wait.until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'provider-info'))) #pagination pagination = driver.find_element(By.XPATH,'//ul[@class="pagination justify-content-center"]') pages = pagination.find_elements(By.TAG_NAME,'li') last_page = int(250) company_names = [] taglines = [] locations = [] costs = [] ratings = [] current_page = 1 while current_page <= last_page: company_elements = wait.until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'provider-info'))) for company_element in company_elements: company_name = company_element.find_element(By.CLASS_NAME, "company_info").text company_names.append(company_name) tagline = company_element.find_element(By.XPATH,'.//p[@class="company_info__wrap tagline"]').text taglines.append(tagline) rating = company_element.find_element(By.XPATH,'.//span[@class="rating sg-rating__number"]').text ratings.append(rating) location = company_element.find_element(By.XPATH, './/span[@class="locality"]').text locations.append(location) cost = company_element.find_element(By.XPATH, './/div[@class="list-item block_tag custom_popover"]').text costs.append(cost) current_page = current_page + 1 try: next_page = driver.find_element(By.XPATH,'//li[@class="page-item next"]/a[@class="page-link"]")') next_page.click() time.sleep(10) except: break driver.close() data = {'Company_Name': company_names, 'Tagline': taglines, 'location': locations, 'Ticket_Price': costs, 'Rating': ratings} df = pd.DataFrame(data) df.to_csv('companies_test1.csv', index=False) print(df)
修复方案及说明
1. 修正XPATH语法错误
原代码中「下一页」按钮的XPATH末尾多了一个多余的双引号,导致元素定位失败,直接触发except分支跳出循环。修正后的定位逻辑需配合智能等待:
next_page = wait.until(EC.element_to_be_clickable((By.XPATH, '//li[@class="page-item next"]/a[@class="page-link"]')))
2. 替换固定等待为智能等待
摒弃time.sleep(10)这类固定时长等待,改用WebDriverWait等待元素可点击/页面状态变化,确保页面完全加载后再执行操作,避免因网络延迟导致的点击失败。
3. 动态获取最后一页页码
不要硬写last_page = 250,从页面分页栏提取实际最后一页数字,适配网站分页结构变化:
# 分页栏最后一个li是「下一页」,取倒数第二个li的文本作为最后一页页码 last_page = int(pagination.find_elements(By.TAG_NAME, 'li')[-2].text)
4. 添加元素定位异常处理
部分公司可能缺失tagline、收费标准等元素,直接定位会导致脚本中断,需给这类步骤加try-except处理:
# 示例:处理缺失标语的情况 try: tagline = company_element.find_element(By.XPATH,'.//p[@class="company_info__wrap tagline"]').text except: tagline = "无标语" taglines.append(tagline)
5. 避免StaleElementReferenceException
每次翻页后页面元素会重新渲染,之前定位的元素会失效,需在循环内重新确认元素状态,或等待旧元素失效后再定位新元素。
修复后的完整代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pandas as pd website = "https://clutch.co/us/web-developers" options = webdriver.ChromeOptions() options.add_experimental_option("detach", False) driver = webdriver.Chrome(options=options) driver.get(website) wait = WebDriverWait(driver, 15) # 初始化存储列表 company_names = [] taglines = [] locations = [] costs = [] ratings = [] # 获取初始分页信息 pagination = wait.until(EC.presence_of_element_located((By.XPATH, '//ul[@class="pagination justify-content-center"]'))) last_page = int(pagination.find_elements(By.TAG_NAME, 'li')[-2].text) current_page = 1 while current_page <= last_page: # 等待当前页公司元素加载完成 company_elements = wait.until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'provider-info'))) for company_element in company_elements: # 公司名称 company_name = company_element.find_element(By.CLASS_NAME, "company_info").text company_names.append(company_name) # 标语(处理缺失情况) try: tagline = company_element.find_element(By.XPATH,'.//p[@class="company_info__wrap tagline"]').text except: tagline = "无标语" taglines.append(tagline) # 评分 rating = company_element.find_element(By.XPATH,'.//span[@class="rating sg-rating__number"]').text ratings.append(rating) # 地区 location = company_element.find_element(By.XPATH, './/span[@class="locality"]').text locations.append(location) # 收费标准(处理缺失情况) try: cost = company_element.find_element(By.XPATH, './/div[@class="list-item block_tag custom_popover"]').text except: cost = "未公示" costs.append(cost) # 翻页逻辑 if current_page < last_page: try: # 等待下一页按钮可点击 next_page = wait.until(EC.element_to_be_clickable((By.XPATH, '//li[@class="page-item next"]/a[@class="page-link"]'))) next_page.click() # 等待下一页加载完成(通过旧元素失效判断) wait.until(EC.staleness_of(company_elements[0])) current_page += 1 except Exception as e: print(f"翻页失败: {str(e)}") break else: break driver.quit() # 保存数据 data = { 'Company_Name': company_names, 'Tagline': taglines, 'Location': locations, 'Cost': costs, 'Rating': ratings } df = pd.DataFrame(data) df.to_csv('clutch_web_developers.csv', index=False) print("数据保存完成,共获取", len(df), "条记录") print(df.head())
内容的提问来源于stack exchange,提问作者Zerfue
相关产品推荐
相关产品推荐

