Selenium爬虫使用Class Selector无法提取网页价格问题求助
问题:Headless模式下Selenium无法定位房产价格元素
在抓取https://www.spiti24.gr/en/for-sale/property/glyfada的房产价格时,启用headless模式后,self.driver.find_elements(By.CLASS_NAME, "property__price")始终返回空列表,完整代码如下:
from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.by import By from selenium.webdriver.common.action_chains import ActionChains from fake_useragent import UserAgent class PropertyScraper: def __init__(self, base_url, location): options = Options() ua = UserAgent() userAgent = ua.random options.add_argument(f'user-agent={userAgent}') options.add_argument("--window-size=1920,1080") options.add_argument("--start-maximized") options.add_argument("--headless") options.add_argument("--disable-gpu") options.add_argument("--no-sandbox") options.add_argument("--disable-dev-shm-usage") self.driver = webdriver.Chrome(options=options) self.driver.implicitly_wait(20) self.driver.get(f"{base_url}{location}") def scrape_property_data(self, pages_to_scrape): all_data = [] for _ in range(pages_to_scrape): property_prices = self.driver.find_elements(By.CLASS_NAME, "property__price") property_square_meters = self.driver.find_elements(By.CLASS_NAME, "property__title__parts") for price_element, square_meter_element in zip(property_prices, property_square_meters): price = price_element.text.strip() square_meter = square_meter_element.text.strip() all_data.append({"price": price, "square_meter": square_meter}) # time.sleep(random.uniform(5, 10)) # Random delay between 5 to 10 seconds next_button = self.driver.find_element(By.CSS_SELECTOR, "li.next.enabled") actions = ActionChains(self.driver) actions.move_to_element(next_button).perform() # Move to the "next" button next_button.click() return all_data def close_driver(self): self.driver.quit() def main(): base_url = "https://www.spiti24.gr/en/for-sale/property/" location = "glyfada" pages_to_scrape = 5 scraper = PropertyScraper(base_url, location) scraped_data = scraper.scrape_property_data(pages_to_scrape) scraper.close_driver() print(scraped_data) if __name__ == "__main__": main()
问题原因排查
- Headless模式特征被检测:旧版
--headless模式的Chrome带有明显识别特征(如UA包含"HeadlessChrome"),网站会返回不同页面结构或阻止内容渲染。 - 元素加载时机未匹配:隐式等待无法覆盖所有动态加载场景,可能元素未渲染完成就执行了定位操作。
- Cookie弹窗未处理:网站的cookie同意弹窗可能遮挡或阻止内容加载,导致元素无法被定位。
- User-Agent有效性不足:随机生成的UA可能不符合网站的浏览器版本要求,被判定为非真实用户。
解决办法
1. 使用新版Headless模式
Chrome 112+支持--headless=new参数,该模式更接近真实Chrome浏览器,大幅降低被检测概率。
2. 优化浏览器配置,模拟真实用户
- 固定真实Chrome的UA(避免随机UA带来的不稳定)
- 添加禁用自动化特征检测的配置,移除
navigator.webdriver属性
3. 显式等待元素加载
使用WebDriverWait等待目标元素出现,比隐式等待更可靠。
4. 处理Cookie弹窗
先定位并点击同意按钮,确保页面内容正常加载。
修改后的完整代码
from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time import random class PropertyScraper: def __init__(self, base_url, location): options = Options() # 使用新版headless模式 options.add_argument("--headless=new") # 固定真实Chrome UA(可根据你的Chrome版本调整) options.add_argument('user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36') options.add_argument("--window-size=1920,1080") options.add_argument("--start-maximized") options.add_argument("--disable-gpu") options.add_argument("--no-sandbox") options.add_argument("--disable-dev-shm-usage") # 禁用自动化检测特征 options.add_argument("--disable-blink-features=AutomationControlled") options.add_experimental_option("excludeSwitches", ["enable-automation"]) options.add_experimental_option('useAutomationExtension', False) self.driver = webdriver.Chrome(options=options) # 移除webdriver属性,避免被检测 self.driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})") self.driver.get(f"{base_url}{location}") # 处理cookie弹窗 try: cookie_btn = WebDriverWait(self.driver, 10).until( EC.element_to_be_clickable((By.CSS_SELECTOR, "button#onetrust-accept-btn-handler")) ) cookie_btn.click() time.sleep(random.uniform(2, 3)) except: # 如果没有弹窗,跳过 pass def scrape_property_data(self, pages_to_scrape): all_data = [] wait = WebDriverWait(self.driver, 20) for _ in range(pages_to_scrape): # 等待价格元素加载完成 property_prices = wait.until( EC.presence_of_all_elements_located((By.CLASS_NAME, "property__price")) ) property_square_meters = wait.until( EC.presence_of_all_elements_located((By.CLASS_NAME, "property__title__parts")) ) for price_element, square_meter_element in zip(property_prices, property_square_meters): price = price_element.text.strip() square_meter = square_meter_element.text.strip() all_data.append({"price": price, "square_meter": square_meter}) # 随机延迟,模拟用户行为 time.sleep(random.uniform(3, 6)) # 等待下一页按钮可点击 next_button = wait.until( EC.element_to_be_clickable((By.CSS_SELECTOR, "li.next.enabled")) ) # 滚动到按钮位置 self.driver.execute_script("arguments[0].scrollIntoView(true);", next_button) time.sleep(random.uniform(1, 2)) next_button.click() # 等待页面加载完成 wait.until(EC.staleness_of(property_prices[0])) return all_data def close_driver(self): self.driver.quit() def main(): base_url = "https://www.spiti24.gr/en/for-sale/property/" location = "glyfada" pages_to_scrape = 5 scraper = PropertyScraper(base_url, location) scraped_data = scraper.scrape_property_data(pages_to_scrape) scraper.close_driver() print(scraped_data) if __name__ == "__main__": main()
关键修改说明
- 替换旧版
--headless为--headless=new,降低被反爬检测的概率 - 添加禁用自动化检测的配置,移除
navigator.webdriver属性 - 新增Cookie弹窗处理逻辑,确保页面内容正常加载
- 使用
WebDriverWait显式等待元素,避免因加载时机问题导致的空结果 - 添加随机延迟和滚动操作,模拟真实用户行为,减少触发反爬机制的风险
内容的提问来源于stack exchange,提问作者asd
相关产品推荐
相关产品推荐

