You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium爬虫使用Class Selector无法提取网页价格问题求助

问题:Headless模式下Selenium无法定位房产价格元素

在抓取https://www.spiti24.gr/en/for-sale/property/glyfada的房产价格时,启用headless模式后,self.driver.find_elements(By.CLASS_NAME, "property__price")始终返回空列表,完整代码如下:

from selenium import webdriver
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.common.by import By
from selenium.webdriver.common.action_chains import ActionChains
from fake_useragent import UserAgent
class PropertyScraper:
    def __init__(self, base_url, location):
        options = Options()
        ua = UserAgent()
        userAgent = ua.random
        options.add_argument(f'user-agent={userAgent}')
        options.add_argument("--window-size=1920,1080")
        options.add_argument("--start-maximized")
        options.add_argument("--headless")
        options.add_argument("--disable-gpu")
        options.add_argument("--no-sandbox")
        options.add_argument("--disable-dev-shm-usage")
        self.driver = webdriver.Chrome(options=options)
        self.driver.implicitly_wait(20)
        self.driver.get(f"{base_url}{location}")

    def scrape_property_data(self, pages_to_scrape):
        all_data = []  
        for _ in range(pages_to_scrape):
            property_prices = self.driver.find_elements(By.CLASS_NAME, "property__price")
            property_square_meters = self.driver.find_elements(By.CLASS_NAME, "property__title__parts")
            for price_element, square_meter_element in zip(property_prices, property_square_meters):
                price = price_element.text.strip()
                square_meter = square_meter_element.text.strip()
                all_data.append({"price": price, "square_meter": square_meter})
            # time.sleep(random.uniform(5, 10))  # Random delay between 5 to 10 seconds
            next_button = self.driver.find_element(By.CSS_SELECTOR, "li.next.enabled")
            actions = ActionChains(self.driver)
            actions.move_to_element(next_button).perform()  # Move to the "next" button
            next_button.click()
        return all_data

    def close_driver(self):
        self.driver.quit()

def main():
    base_url = "https://www.spiti24.gr/en/for-sale/property/"
    location = "glyfada"
    pages_to_scrape = 5  
    scraper = PropertyScraper(base_url, location)
    scraped_data = scraper.scrape_property_data(pages_to_scrape)
    scraper.close_driver()
    print(scraped_data)

if __name__ == "__main__":
    main()

问题原因排查

  • Headless模式特征被检测:旧版--headless模式的Chrome带有明显识别特征(如UA包含"HeadlessChrome"),网站会返回不同页面结构或阻止内容渲染。
  • 元素加载时机未匹配:隐式等待无法覆盖所有动态加载场景,可能元素未渲染完成就执行了定位操作。
  • Cookie弹窗未处理:网站的cookie同意弹窗可能遮挡或阻止内容加载,导致元素无法被定位。
  • User-Agent有效性不足:随机生成的UA可能不符合网站的浏览器版本要求,被判定为非真实用户。

解决办法

1. 使用新版Headless模式

Chrome 112+支持--headless=new参数,该模式更接近真实Chrome浏览器,大幅降低被检测概率。

2. 优化浏览器配置,模拟真实用户

  • 固定真实Chrome的UA(避免随机UA带来的不稳定)
  • 添加禁用自动化特征检测的配置,移除navigator.webdriver属性

3. 显式等待元素加载

使用WebDriverWait等待目标元素出现,比隐式等待更可靠。

4. 处理Cookie弹窗

先定位并点击同意按钮,确保页面内容正常加载。

修改后的完整代码

from selenium import webdriver
from selenium.webdriver.chrome.options import Options
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import time
import random

class PropertyScraper:
    def __init__(self, base_url, location):
        options = Options()
        # 使用新版headless模式
        options.add_argument("--headless=new")
        # 固定真实Chrome UA(可根据你的Chrome版本调整)
        options.add_argument('user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36')
        options.add_argument("--window-size=1920,1080")
        options.add_argument("--start-maximized")
        options.add_argument("--disable-gpu")
        options.add_argument("--no-sandbox")
        options.add_argument("--disable-dev-shm-usage")
        # 禁用自动化检测特征
        options.add_argument("--disable-blink-features=AutomationControlled")
        options.add_experimental_option("excludeSwitches", ["enable-automation"])
        options.add_experimental_option('useAutomationExtension', False)
        
        self.driver = webdriver.Chrome(options=options)
        # 移除webdriver属性,避免被检测
        self.driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})")
        self.driver.get(f"{base_url}{location}")
        # 处理cookie弹窗
        try:
            cookie_btn = WebDriverWait(self.driver, 10).until(
                EC.element_to_be_clickable((By.CSS_SELECTOR, "button#onetrust-accept-btn-handler"))
            )
            cookie_btn.click()
            time.sleep(random.uniform(2, 3))
        except:
            # 如果没有弹窗,跳过
            pass

    def scrape_property_data(self, pages_to_scrape):
        all_data = []  
        wait = WebDriverWait(self.driver, 20)
        for _ in range(pages_to_scrape):
            # 等待价格元素加载完成
            property_prices = wait.until(
                EC.presence_of_all_elements_located((By.CLASS_NAME, "property__price"))
            )
            property_square_meters = wait.until(
                EC.presence_of_all_elements_located((By.CLASS_NAME, "property__title__parts"))
            )
            
            for price_element, square_meter_element in zip(property_prices, property_square_meters):
                price = price_element.text.strip()
                square_meter = square_meter_element.text.strip()
                all_data.append({"price": price, "square_meter": square_meter})
            
            # 随机延迟,模拟用户行为
            time.sleep(random.uniform(3, 6))
            
            # 等待下一页按钮可点击
            next_button = wait.until(
                EC.element_to_be_clickable((By.CSS_SELECTOR, "li.next.enabled"))
            )
            # 滚动到按钮位置
            self.driver.execute_script("arguments[0].scrollIntoView(true);", next_button)
            time.sleep(random.uniform(1, 2))
            next_button.click()
            # 等待页面加载完成
            wait.until(EC.staleness_of(property_prices[0]))
        return all_data

    def close_driver(self):
        self.driver.quit()

def main():
    base_url = "https://www.spiti24.gr/en/for-sale/property/"
    location = "glyfada"
    pages_to_scrape = 5  
    scraper = PropertyScraper(base_url, location)
    scraped_data = scraper.scrape_property_data(pages_to_scrape)
    scraper.close_driver()
    print(scraped_data)

if __name__ == "__main__":
    main()

关键修改说明

  • 替换旧版--headless为--headless=new,降低被反爬检测的概率
  • 添加禁用自动化检测的配置,移除navigator.webdriver属性
  • 新增Cookie弹窗处理逻辑,确保页面内容正常加载
  • 使用WebDriverWait显式等待元素,避免因加载时机问题导致的空结果
  • 添加随机延迟和滚动操作,模拟真实用户行为,减少触发反爬机制的风险

内容的提问来源于stack exchange,提问作者asd

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.15 01:45:59