如何实现Python商品卡片多页自动解析?已完成单页解析
商品卡片多页解析解决方案
先给你两个可行的方案,优先推荐第一种(不用Selenium),效率更高;如果网站是动态分页,再用第二种。
方案一:直接通过URL分页参数遍历(静态分页)
很多电商网站的分页靠URL里的page参数实现,比如第一页是?category=926,第二页就是?category=926&page=2,以此类推。你可以循环遍历页码,直到页面没有商品为止。
修改后的完整代码:
import random import string import csv import requests from bs4 import BeautifulSoup base_url = "https://game29.ru/products?category=926&page={}" page_num = 1 all_products = [] # 固定字段,无需重复定义 ad_status = "Free" category = "Игры, приставки и программы" goods_type = "Игры для приставок" ad_type = "Продаю своё" adress = "" discription = "" condition = "Новое" data_begin = "2024-04-03" data_end = "2024-05-03" allow_email = "Нет" contact_phone = "" contact_method = "По телефону и в сообщениях" multi_class = {'class': ['row'], 'style': 'border: 2px solid #898989;border-radius: 7px;padding: 2px;margin-top: -2px;'} while True: url = base_url.format(page_num) response = requests.get(url) # 检查请求是否成功 if response.status_code != 200: print(f"第{page_num}页请求失败,状态码:{response.status_code}") break soup = BeautifulSoup(response.text, "html.parser") products = soup.find_all("div", {"class":"row"}) # 标记当前页是否有有效商品 has_products = False for product in products: if product.attrs == multi_class: has_products = True # 每个商品生成唯一ID(之前的代码里只生成了一次,现在移到循环内) identifaer = "".join([random.choice(string.ascii_letters + string.digits) for n in range(32)]) image ="https://www.game29.ru" + product.find("img")["src"] if image != "https://game29.ru/zaglushka.png": title = product.find("div", {"class":"cart-item-name"}).text price = product.find("div", {"class": "cart-item-price"}).text.strip().replace("руб.", "") all_products.append([identifaer, ad_status, category, goods_type, ad_type, adress, title, discription, condition, price, data_begin, data_end, allow_email, contact_phone, image, contact_method]) # 无商品则终止循环 if not has_products: print(f"第{page_num}页无商品,结束爬取") break print(f"已完成第{page_num}页爬取") page_num += 1 # 写入CSV文件 names = ["Id", "AdStatus", "Category", "GoodsType", "Adtype", "Adress", "Title", "Discription", "Condition", "Price", "DataBegin", "DataEnd", "AllowEmail", "ContactPhone","ImageUrls", "ContactMethod"] with open("data.csv", "w", newline='', encoding='utf-8') as csv_file: writer = csv.writer(csv_file, delimiter=',') writer.writerow(names) writer.writerows(all_products) print(f"爬取完成,共{len(all_products)}条商品数据")
关键调整说明:
- 把唯一ID生成逻辑移到商品循环内,保证每个商品ID不重复
- 加入请求状态码检查,避免无效请求浪费资源
- 用
while True循环自动遍历页码,直到无商品时停止 - CSV写入改用
w模式(覆盖生成),避免多次运行重复追加数据
方案二:用Selenium处理动态分页(JS加载页面)
如果网站的分页是点击后通过AJAX动态加载(不刷新整个页面),就需要用Selenium模拟浏览器操作。
步骤1:安装依赖
pip install selenium
同时下载对应浏览器的驱动(比如ChromeDriver,需与浏览器版本匹配),放到项目目录或系统PATH中。
完整代码:
import random import string import csv from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from bs4 import BeautifulSoup # 固定字段 ad_status = "Free" category = "Игры, приставки и программы" goods_type = "Игры для приставок" ad_type = "Продаю своё" adress = "" discription = "" condition = "Новое" data_begin = "2024-04-03" data_end = "2024-05-03" allow_email = "Нет" contact_phone = "" contact_method = "По телефону и в сообщениях" multi_class = {'class': ['row'], 'style': 'border: 2px solid #898989;border-radius: 7px;padding: 2px;margin-top: -2px;'} all_products = [] # 初始化Chrome浏览器 driver = webdriver.Chrome() driver.get("https://game29.ru/products?category=926") while True: # 获取当前页面HTML源码 html = driver.page_source soup = BeautifulSoup(html, "html.parser") products = soup.find_all("div", {"class":"row"}) has_products = False for product in products: if product.attrs == multi_class: has_products = True identifaer = "".join([random.choice(string.ascii_letters + string.digits) for n in range(32)]) image ="https://www.game29.ru" + product.find("img")["src"] if image != "https://game29.ru/zaglushka.png": title = product.find("div", {"class":"cart-item-name"}).text price = product.find("div", {"class": "cart-item-price"}).text.strip().replace("руб.", "") all_products.append([identifaer, ad_status, category, goods_type, ad_type, adress, title, discription, condition, price, data_begin, data_end, allow_email, contact_phone, image, contact_method]) if not has_products: break # 尝试点击下一页按钮(需根据页面实际元素调整选择器) try: # 等待下一页按钮可点击(假设按钮文本为"Следующая") next_button = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, "//a[text()='Следующая']")) ) # 检查是否为最后一页(按钮可能带有disabled类) if "disabled" in next_button.get_attribute("class"): break next_button.click() # 等待页面加载完成(等待商品列表元素更新) WebDriverWait(driver, 10).until( EC.staleness_of(driver.find_element(By.CLASS_NAME, "row")) ) except Exception as e: print(f"无法点击下一页:{e}") break # 关闭浏览器 driver.quit() # 写入CSV names = ["Id", "AdStatus", "Category", "GoodsType", "Adtype", "Adress", "Title", "Discription", "Condition", "Price", "DataBegin", "DataEnd", "AllowEmail", "ContactPhone","ImageUrls", "ContactMethod"] with open("data.csv", "w", newline='', encoding='utf-8') as csv_file: writer = csv.writer(csv_file, delimiter=',') writer.writerow(names) writer.writerows(all_products) print(f"爬取完成,共{len(all_products)}条商品数据")
关键说明:
- 下一页按钮的选择器需根据页面实际HTML调整(比如查看按钮的class或文本)
- 加入等待机制,确保页面加载完成后再解析数据
- 如果网站有反爬机制,可添加随机等待时间、更换User-Agent等优化
内容的提问来源于stack exchange,提问作者Алексей Ржавин
相关产品推荐
相关产品推荐

