Python Selenium爬取房产网站时无法点击元素跳转获取详情问题
Selenium 房产网站爬取无法点击跳转问题修复方案
原代码存在以下核心问题:
- xpath语法错误:代码中残留HTML转义字符
",需替换为实际双引号;找子元素时开头使用//会全局匹配而非在当前定位的房源元素下匹配,导致定位到错误元素 - 未定义变量:提取房源链接的代码中直接使用了未声明的
url变量,直接运行会报错 - 方法弃用:高版本Selenium已废弃
find_element_by_xpath、switch_to_window这类方法,需使用官方推荐的新API - 窗口管理混乱:点击打开新标签页后没有及时关闭,后续
window_handles[1]的索引会错位,导致切换到错误页面 - 点击失效风险:元素可能被遮挡、未完全加载就执行点击,容易触发点击失败异常
- 硬编码等待可靠性差:固定
time.sleep容易受网络波动影响,推荐用显式等待替代
修复后可运行代码
from selenium import webdriver import time from selenium.webdriver.support.wait import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.common.exceptions import TimeoutException, NoSuchElementException from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.by import By Link = 'https://www.altamirarealestate.com.cy/results/for-sale/flats/cyprus/35p009679979327046l33p17435142059772z9' # 初始化浏览器 driver = webdriver.Chrome() driver.maximize_window() wait = WebDriverWait(driver, 10) # 打开目标页面 driver.get(Link) # 接受cookie try: accept_cookie_btn = wait.until(EC.element_to_be_clickable((By.ID, 'onetrust-accept-btn-handler'))) accept_cookie_btn.click() except TimeoutException: print("未找到cookie确认按钮,跳过") # 加载全部房源 while True: try: load_more_btn = wait.until(EC.element_to_be_clickable((By.XPATH, "//*[contains(@value,'View more')]"))) # 用js点击避免被遮挡 driver.execute_script("arguments[0].click();", load_more_btn) time.sleep(3) except TimeoutException: print('所有房源已加载完成') break # 获取所有房源卡片 properties_list = driver.find_elements(By.XPATH, '//*[@class="minificha "]') print(f"共获取到{len(properties_list)}个房源") main_window_handle = driver.current_window_handle property_details = [] for i in range(len(properties_list)): driver.switch_to.window(main_window_handle) current_property = properties_list[i] # 在当前房源卡片下找链接,xpath开头加.表示当前节点下查找 property_link = current_property.find_element(By.XPATH, './/a') # 用js点击避免跳转失败 driver.execute_script("arguments[0].click();", property_link) time.sleep(2) # 切换到房源详情页 for handle in driver.window_handles: if handle != main_window_handle: property_window = handle break driver.switch_to.window(property_window) # 这里可自行补充提取房源详情的逻辑 # 示例:提取房源标题 try: title = driver.find_element(By.XPATH, '//h1').text.strip() print(f"第{i+1}个房源标题:{title}") except NoSuchElementException: print(f"第{i+1}个房源标题提取失败") # 如果需要提取详情页下的公寓列表,参考同样逻辑处理即可: # number_of_flats = driver.find_elements(By.XPATH, './/*[@class="lineainmu "]') # 逐个跳转提取信息后关闭对应标签页即可 # 关闭当前详情页,避免窗口太多索引混乱 driver.close() driver.switch_to.window(main_window_handle) time.sleep(1) # 全部提取完成后关闭浏览器 driver.quit()
内容的提问来源于stack exchange,提问作者Monica Odysseos CY
相关产品推荐
相关产品推荐

