Selenium爬取kotsovolos站点手机数据无法实现翻页导航问题求助
爬虫翻页问题修复方案
核心问题
- 无翻页终止逻辑:无限循环未设置退出条件,到达最后一页时下一页按钮隐藏,元素查找直接报错导致程序终止
- Selenium API不兼容:代码使用的
find_element_by_css_selector是Selenium 3.x的废弃API,4.x版本运行直接抛出异常 - 页面等待逻辑缺失:点击下一页后仅固定等待3秒,未验证页面是否完成跳转,容易出现元素未加载的报错
- 数据未持久化:仅初始化了Excel表头,爬取到的所有商品数据未写入文件,最终导出文件只有表头
修复后代码
import xlwt from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import re import time from datetime import date class kotsovolosmobiles: def __init__(self): self.url='https://www.kotsovolos.gr/mobile-phones-gps/mobile-phones/smartphones?pageSize=60' self.country='GR' self.currency='euro' self.VAT= 'Included' self.shipping = 'Available for shipment' self.Pre_PromotionPrice ='N/A' def kotsovolos(self): # 初始化Excel,修正重复写第一列的错误 wb = xlwt.Workbook() ws = wb.add_sheet('Sheet1',cell_overwrite_ok=True) headers = ["Product_Manufacturer","Product_Url","Product_Price","Product_Model","Memory","Currency","Color","VAT","Shipping Cost","Pre-PromotionPrice","Country","Date","Raw_Model"] for col, header in enumerate(headers): ws.write(0, col, header) driver=webdriver.Chrome() driver.get(self.url) today = date.today() # 等待Cookie弹窗出现并点击 WebDriverWait(driver, 15).until(EC.element_to_be_clickable((By.ID, 'CybotCookiebotDialogBodyLevelButtonLevelOptinAllowAll'))).click() print("cookies accepted") driver.maximize_window() titles = [] models = [] memorys = [] prod_prices = [] p_links =[] p_colors = [] while True: # 等待当前页商品加载完成 storage_box = WebDriverWait(driver, 10).until(EC.presence_of_all_elements_located((By.CSS_SELECTOR, 'div[class="product"]'))) for storage_boxes in storage_box: product_url = storage_boxes.find_element(By.CSS_SELECTOR, 'div[class="title"] a').get_attribute('href') print(product_url) p_links.append(product_url) p_model = storage_boxes.find_element(By.CSS_SELECTOR, 'div[class="title"] a').text print(p_model) models.append(p_model) manufacturer1 = p_model.split(" ") print(manufacturer1[0]) titles.append(manufacturer1[0]) memory = re.findall('\d+ ?[gG][bB]',p_model) print(memory) memory1 = str(memory).replace("['",'').replace("']",'').replace("[]",'').strip() if "," in memory1: arr=memory1.split(",") memory_str = 'N/A' for str1 in arr: str2=str1.replace("GB", "").replace("gb", "").replace("'", "").strip() if len(str2)!=1: memory_str=str1 break elif (memory1 == ""): memory_str ='N/A' else: memory_str=memory1 memory_str = memory_str.replace("'", "").strip() print(memory_str) memorys.append(memory_str) prod_color = p_model.split(" ") length = len(prod_color) indexcolor = length-3 if length >=3 else 0 color1 = prod_color[indexcolor].replace("['",'').replace("']",'').strip() print(color1) p_colors.append(color1) p_price = storage_boxes.find_element(By.CSS_SELECTOR, '.priceWithVat > .price').text print(p_price) prod_prices.append(p_price) # 尝试点击下一页,不存在则退出循环 try: next_btn = WebDriverWait(driver, 5).until(EC.element_to_be_clickable((By.CSS_SELECTOR, '.pagination_next a'))) next_btn.click() print("next page") # 等待当前页商品失效,确认页面跳转完成 WebDriverWait(driver, 10).until(EC.staleness_of(storage_box[0])) except: print("已到最后一页,爬取结束") break # 所有数据写入Excel for row in range(len(titles)): ws.write(row+1, 0, titles[row]) ws.write(row+1, 1, p_links[row]) ws.write(row+1, 2, prod_prices[row]) ws.write(row+1, 3, models[row]) ws.write(row+1, 4, memorys[row]) ws.write(row+1, 5, self.currency) ws.write(row+1, 6, p_colors[row]) ws.write(row+1, 7, self.VAT) ws.write(row+1, 8, self.shipping) ws.write(row+1, 9, self.Pre_PromotionPrice) ws.write(row+1, 10, self.country) ws.write(row+1, 11, str(today)) ws.write(row+1, 12, models[row]) # 保存文件 wb.save(r"C:\Users\Karthick R\Desktop\VS code\kotsovolos.xls") driver.quit() if __name__ == '__main__': kotsovolos_gr = kotsovolosmobiles() kotsovolos_gr.kotsovolos()
运行前置依赖
安装对应依赖包即可运行:
- selenium 4.x版本:
pip install selenium - xlwt:
pip install xlwt - 若使用selenium 4.6以下版本,需手动配置和本地Chrome浏览器版本匹配的ChromeDriver
内容的提问来源于stack exchange,提问作者Vinitha
相关产品推荐
相关产品推荐

