Selenium爬取SPAR匈牙利站仅获6条商品数据及分页失效求助
问题描述
我尝试从Spar匈牙利官网的卫生纸分类页抓取数据,需要获取每个商品的名称、价格及单价。但运行代码后只能拿到36条商品中的6条,同时分页功能也未实现,求解决办法。
原代码
from selenium import webdriver from selenium.webdriver.common.keys import Keys from selenium.webdriver.common.by import By from selenium.webdriver.common.action_chains import ActionChains import time import requests from csv import writer from datetime import date from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC driver = webdriver.Chrome() driver.implicitly_wait(10) url = "https://www.spar.hu/onlineshop/haztartas-es-szabadido/wc-furdoszoba/toalettpapir/c/H12-3-1/" driver.get(url) driver.implicitly_wait(10) webdriver.Chrome().get_cookies() driver.implicitly_wait(10) with open('Spar Toalett papír.csv', 'a', encoding='utf-8', newline='') as f: thewriter = writer(f) header = ['Termék', 'Termék ár', 'Egységár'] thewriter.writerow(header) for i in range(1): driver.implicitly_wait(15) items = WebDriverWait(driver, 10).until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'productBox'))) items = driver.find_elements(By.CLASS_NAME, "productBoxTop.cf.j-toggleProductInfo") print(len(items)) driver.implicitly_wait(50) for item in items: driver.implicitly_wait(50) name = item.find_element(By.CLASS_NAME, 'productTitle').text driver.execute_script("window.scrollTo(0,100);") price = item.find_element(By.CLASS_NAME, 'priceInteger').text unit = item.find_element(By.CLASS_NAME, 'extraInfoPrice').text #info = [id, datetoday, Sorszam, bolt, name, price] thewriter.writerow(info) print(sorszam, name, price, unit)
解决方案
一、问题根源分析
- 页面滚动加载机制:初始页面仅显示少量商品,需滚动到底部触发剩余商品加载
- 无效的
get_cookies()调用:webdriver.Chrome().get_cookies()会新建一个独立的Chrome实例,导致当前操作的浏览器上下文丢失 - 元素定位与变量缺失:选择器匹配范围有限,且
info、sorszam等变量未定义,导致数据写入失败 - 无分页逻辑:未处理分页按钮的点击与页面切换
二、修复后完整代码(含滚动加载+分页)
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time from csv import writer from datetime import date # 初始化浏览器 driver = webdriver.Chrome() driver.implicitly_wait(10) target_url = "https://www.spar.hu/onlineshop/haztartas-es-szabadido/papiraru/toalettpapir/c/H12-1-2/" driver.get(target_url) # 准备CSV文件(按日期命名避免覆盖) today = date.today().strftime("%Y-%m-%d") with open(f'Spar_Toalett_papir_{today}.csv', 'w', encoding='utf-8', newline='') as f: thewriter = writer(f) header = ['序号', '日期', '商品名称', '商品价格', '单价'] thewriter.writerow(header) item_total = 0 while True: # 滚动到底部加载所有商品 last_page_height = driver.execute_script("return document.body.scrollHeight") while True: driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(2) # 等待加载完成 new_page_height = driver.execute_script("return document.body.scrollHeight") if new_page_height == last_page_height: break # 无新内容加载,退出滚动 last_page_height = new_page_height # 获取所有商品容器 items = WebDriverWait(driver, 15).until( EC.presence_of_all_elements_located((By.CLASS_NAME, 'productBox')) ) print(f"当前页面加载完成,共找到 {len(items)} 个商品") # 遍历提取商品数据 for item in items: item_total += 1 try: name = item.find_element(By.CLASS_NAME, 'productTitle').text # 提取完整价格(整数+小数部分) price_int = item.find_element(By.CLASS_NAME, 'priceInteger').text price_dec = item.find_element(By.CLASS_NAME, 'priceDecimal').text full_price = f"{price_int},{price_dec} Ft" unit_price = item.find_element(By.CLASS_NAME, 'extraInfoPrice').text # 写入CSV row_data = [item_total, today, name, full_price, unit_price] thewriter.writerow(row_data) print(f"{item_total} | {name} | {full_price} | {unit_price}") except Exception as e: print(f"提取商品数据失败: {str(e)}") continue # 处理分页:查找可点击的下一页按钮 try: next_btn = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.CSS_SELECTOR, 'a.paginationNext:not(.disabled)')) ) next_btn.click() time.sleep(3) # 等待分页页面加载 except: print("已到达最后一页,抓取结束") break driver.quit()
三、关键修复说明
- 移除无效代码:删除
webdriver.Chrome().get_cookies(),避免新建独立浏览器实例 - 滚动加载实现:通过循环滚动页面并对比高度,确保所有商品加载完成
- 完整价格提取:补充价格小数部分的抓取,还原真实商品价格
- 分页逻辑:通过CSS选择器判断下一页按钮是否可用,实现自动翻页
- 变量补全与错误捕获:定义缺失的序号、日期等变量,添加try-except避免单个商品提取失败导致程序中断
- CSV优化:按日期命名文件,避免重复写入覆盖数据
内容的提问来源于stack exchange,提问作者Bazsi
相关产品推荐
相关产品推荐

