You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

求助:解决Catawiki页面爬取受阻问题(附Selenium代码)

问题描述

无法爬取烈酒分类页面及后续分页,怀疑是Cookie弹窗或反爬机制导致,附上当前使用的Selenium代码,请求实现商品标题和价格的爬取:

import csv
import time
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from webdriver_manager.chrome import ChromeDriverManager

# -------------------- SETUP SELENIUM --------------------
service = Service(ChromeDriverManager().install())
driver = webdriver.Chrome(service=service)
wait = WebDriverWait(driver, 10)  # Explicit wait for elements

# -------------------- FUNCTION TO CLICK COOKIE POPUP --------------------
def accept_cookies():
    try:
        # Wait for the cookie button & click (Corrected CSS Selector)
        cookie_button = wait.until(
            EC.element_to_be_clickable((By.CSS_SELECTOR, ".c-button.c-button--primary"))
        )
        cookie_button.click()
        print("✅ Cookies accepted!")

        # ⏳ Add short delay after clicking to let the page adjust
        time.sleep(3)

    except Exception:
        print("⚠️ No cookie popup found or already accepted.")

# -------------------- PREPARE CSV FILE --------------------
csv_filename = "cata.csv"

with open(csv_filename, mode="w", newline="", encoding="utf-8") as file:
    writer = csv.writer(file)
    writer.writerow(["Product Name", "Price"])  # Write header

    # -------------------- LOOP THROUGH PAGES --------------------
    for page_num in range(1, 2):  # Pages 1 to 13
        url = f"https://www.catawiki.com/fr/c/965-rhum-cognac-et-spiritueux?page={page_num}"
        driver.get(url)

        # ✅ Accept cookies on every page
        accept_cookies()

        try:
            # ⏳ Wait until products are fully loaded before scraping
            name_elements = wait.until(
                EC.presence_of_all_elements_located((By.CSS_SELECTOR, ".u-typography-h6.c-lot-card__title.u-truncate-2-lines"))
            )

            price_elements = wait.until(
                EC.presence_of_all_elements_located((By.CSS_SELECTOR, ".u-typography-h5.c-lot-card__price"))
            )

            # Extract text and store in CSV
            for name, price in zip(name_elements, price_elements):
                writer.writerow([name.text.strip(), price.text.strip()])

            print(f"✅ Page {page_num} scraped successfully!")

        except Exception as e:
            print(f"⚠️ Error on page {page_num}: {e}")

        # 🔄 Short delay before moving to the next page
        time.sleep(2)

# Close the browser
driver.quit()

print(f"\n✅ Data successfully saved to {csv_filename}!")
解决方案

下面是优化后的代码,解决反爬检测、Cookie弹窗定位不稳定等问题:

import csv
import time
from selenium import webdriver
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from webdriver_manager.chrome import ChromeDriverManager
from selenium.webdriver.chrome.options import Options

# -------------------- 配置Chrome,绕过反爬检测 --------------------
chrome_options = Options()
# 设置模拟真实浏览器的User-Agent
chrome_options.add_argument("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36")
# 禁用自动化扩展和提示
chrome_options.add_argument("--disable-extensions")
chrome_options.add_argument("--disable-popup-blocking")
# 隐藏Selenium自动化标识
chrome_options.add_experimental_option("excludeSwitches", ["enable-automation"])
chrome_options.add_experimental_option('useAutomationExtension', False)
# 启用无痕模式(可选,减少缓存影响)
chrome_options.add_argument("--incognito")

# -------------------- 初始化浏览器 --------------------
service = Service(ChromeDriverManager().install())
driver = webdriver.Chrome(service=service, options=chrome_options)
wait = WebDriverWait(driver, 15)  # 延长等待时间应对慢加载

# -------------------- 更可靠的Cookie弹窗处理 --------------------
def accept_cookies():
    try:
        # 先等待Cookie弹窗容器加载,再定位按钮
        cookie_container = wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, ".c-cookie-banner")))
        cookie_button = wait.until(
            EC.element_to_be_clickable((By.CSS_SELECTOR, ".c-cookie-banner .c-button.c-button--primary"))
        )
        cookie_button.click()
        print("✅ Cookies accepted!")
        time.sleep(2)
    except Exception:
        print("⚠️ No cookie popup found or already accepted.")

# -------------------- 准备CSV文件 --------------------
csv_filename = "cata_spirits.csv"

with open(csv_filename, mode="w", newline="", encoding="utf-8") as file:
    writer = csv.writer(file)
    writer.writerow(["商品名称", "价格"])  # 改为中文表头

    # -------------------- 循环爬取分页 --------------------
    for page_num in range(1, 14):  # 爬取1到13页
        url = f"https://www.catawiki.com/fr/c/965-rhum-cognac-et-spiritueux?page={page_num}"
        driver.get(url)
        
        # 处理Cookie弹窗
        accept_cookies()

        try:
            # 等待商品元素可见(比presence更可靠,确保元素加载完成)
            name_elements = wait.until(
                EC.visibility_of_all_elements_located((By.CSS_SELECTOR, ".c-lot-card__title.u-typography-h6"))
            )
            price_elements = wait.until(
                EC.visibility_of_all_elements_located((By.CSS_SELECTOR, ".c-lot-card__price.u-typography-h5"))
            )

            # 提取数据并写入CSV
            for name, price in zip(name_elements, price_elements):
                writer.writerow([name.text.strip(), price.text.strip()])

            print(f"✅ 第{page_num}页爬取成功!")

        except Exception as e:
            print(f"⚠️ 第{page_num}页爬取失败:{str(e)}")
            # 失败后截图留存(可选)
            driver.save_screenshot(f"error_page_{page_num}.png")

        # 随机延迟,模拟人类浏览行为
        time.sleep(2 + (page_num % 3))  # 2-4秒随机延迟

# 关闭浏览器
driver.quit()

print(f"\n✅ 数据已成功保存到 {csv_filename}!")

关键优化点说明

  • 反爬绕过:添加User-Agent、隐藏Selenium自动化标识,避免被网站检测为爬虫
  • Cookie弹窗处理:先定位弹窗容器再找按钮,提升定位稳定性
  • 元素等待:改用visibility_of_all_elements_located,确保元素完全可见后再提取,避免空文本
  • 人性化延迟:添加随机延迟,模拟真实用户浏览节奏,降低被封风险
  • 异常处理:添加截图功能,方便排查页面加载异常问题

内容的提问来源于stack exchange,提问作者Yo'python

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.14 01:04:53