You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何抓取realtor.com多页房源数据并解决Selenium元素定位报错问题

问题描述

使用Beautiful Soup编写代码按邮政编码爬取realtor.com的房源列表数据,仅能拉取到第一页的47条房源信息,无法获取后续分页的数据。测试用邮政编码94016对应约2000条房源,页面底部有「next」跳转按钮,先后尝试Selenium模拟点击跳转、直接拼接页码到URL后批量请求两种方案,其中Selenium出现元素定位报错,需要更高效的方案拉取全部房源数据。

已尝试的代码实现

初始单页测试代码

from bs4 import BeautifulSoup
import requests

headers = {'User-Agent':'Mozilla/5.0 (Windows NT 6.1) ' \
                                  'AppleWebKit/537.36 (KHTML, like Gecko) ' \
                                  'Chrome/88.0.4324.150 Safari/537.36',
'Accept-Encoding': 'identity'
}

url = 'https://www.realtor.com/realestateandhomes-search/94016'

response=requests.get(url,headers=headers)

soup=BeautifulSoup(response.content,'lxml')
price_list = []

for item in soup.select('.component_property-card'):
    try:
        print('**********')
        print(item.select('[data-label=pc-price]')[0].get_text())
        print(item.select('img')[0]['data-src'])
        print(item.select('.summary-wrap')[0].get_text())
        print(item.select('.address')[0].get_text())
        print(item.select('.property-meta')[0].get_text())
        print(item.select('.special-feature-list')[0].get_text())
        
        price_list.append(item.select('[data-label=pc-price]')[0].get_text())
        
    except Exception as e:
        print('')

拼接URL分页测试代码

# first run
soup_list=[]
import time
import numpy as np
from bs4 import BeautifulSoup
import requests

headers = {'User-Agent':'Mozilla/5.0 (Windows NT 6.1) ' \
                                  'AppleWebKit/537.36 (KHTML, like Gecko) ' \
                                  'Chrome/88.0.4324.150 Safari/537.36',
'Accept-Encoding': 'identity'
}

url = 'https://www.realtor.com/realestateandhomes-search/94016'

response=requests.get(url,headers=headers)
soup=BeautifulSoup(response.content,'lxml')
i=2
print(url)
print('lenght: '+str(len(soup.select('.component_property-card')[0])))
print(str(i))

while len(soup.select('.component_property-card'))!=0:
    try:
        headers = {'User-Agent':'Mozilla/5.0 (Windows NT 6.1) ' \
                                      'AppleWebKit/537.36 (KHTML, like Gecko) ' \
                                      'Chrome/88.0.4324.150 Safari/537.36',
        'Accept-Encoding': 'identity'
        }
        # waits between pulling data
        time.sleep(np.random.randint(low=60, high=70, size=1)[0])
        url = 'https://www.realtor.com/realestateandhomes-search/94016'+'/pg-'+str(i)
        print(url)
        response=requests.get(url,headers=headers)
        soup=BeautifulSoup(response.content,'lxml')
        print('length: '+str(len(soup.select('.component_property-card')[0])))
        i=i+1
        print(str(i))
        soup_list.append(soup)
        
    except:
        headers = {'User-Agent':'Mozilla/5.0 (Windows NT 6.1) ' \
                                      'AppleWebKit/537.36 (KHTML, like Gecko) ' \
                                      'Chrome/88.0.4324.150 Safari/537.36',
        'Accept-Encoding': 'identity'
        }
        # waits between pulling data
        time.sleep(np.random.randint(low=60, high=70, size=1)[0])
        url = 'https://www.realtor.com/realestateandhomes-search/94016'+'/pg-'+str(i)
        print(url)
        response=requests.get(url,headers=headers)
        soup=BeautifulSoup(response.content,'lxml')
        print('length: '+str(len(soup.select('.component_property-card')[0])))
        i=i+1
        print(str(i))
        soup_list.append(soup)

Selenium模拟点击测试代码

from selenium import webdriver
from bs4 import BeautifulSoup
from time import sleep
import os
from selenium.webdriver.chrome.options import Options

driver = webdriver.Chrome(executable_path=os.path.abspath("chromedriver"))

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/92.0.4515.131 Safari/537.3"
}

url = 'https://www.realtor.com/realestateandhomes-search/94534'
driver.get(url)
sleep(3)

# dictionary to store page title as key and html scraped as value
pages = {}
soup = BeautifulSoup(driver.page_source, "html.parser")
pages['Page 1'] = soup

for i in range(0, 4):
    driver.find_element_by_xpath('//*[@id="srp-body"]/section[1]/div[2]/div/a[8]').click()
    sleep(5)
    soup = BeautifulSoup(driver.page_source, "html.parser")
    pages[soup.find('title').text.split('|')[0].strip()] = soup
    
driver.close()

print(len(pages), '\n')
print(pages.keys(), '\n\n')
for k, v in pages.items():
    # just as a check print the first house address for each page
    print(v.find('div', class_ =\
                 'jsx-303111361 address ellipsis srp-page-address srp-address-redesign').text, '\n')
Selenium运行报错信息
---------------------------------------------------------------------------
NoSuchElementException                    Traceback (most recent call last)
<ipython-input-3-b50566b8c5f4> in <module>()
     45 
     46 for i in range(0, 4):
---> 47     driver.find_element_by_xpath('//*[@id="srp-body"]/section[1]/div[2]/div/a[8]').click()
     48     sleep(5)
     49     soup = BeautifulSoup(driver.page_source, "html.parser")

~/anaconda/envs/py36/lib/python3.6/site-packages/selenium/webdriver/remote/webdriver.py in find_element_by_xpath(self, xpath)
    391             element = driver.find_element_by_xpath('//div/td[1]')
    392         """
---> 393         return self.find_element(by=By.XPATH, value=xpath)
    394 
    395     def find_elements_by_xpath(self, xpath):

~/anaconda/envs/py36/lib/python3.6/site-packages/selenium/webdriver/remote/webdriver.py in find_element(self, by, value)
    964         return self.execute(Command.FIND_ELEMENT, {
    965             'using': by,
---> 966             'value': value})['value']
    967 
    968     def find_elements(self, by=By.ID, value=None):

~/anaconda/envs/py36/lib/python3.6/site-packages/selenium/webdriver/remote/webdriver.py in execute(self, driver_command, params)
    318         response = self.command_executor.execute(driver_command, params)
    319         if response:
---> 320             self.error_handler.check_response(response)
    321             response['value'] = self._unwrap_value(
    322                 response.get('value', None))

~/anaconda/envs/py36/lib/python3.6/site-packages/selenium/webdriver/remote/errorhandler.py in check_response(self, response)
    240                 alert_text = value['alert'].get('text')
    241             raise exception_class(message, screen, stacktrace, alert_text)
---> 242         raise exception_class(message, screen, stacktrace)
    243 
    244     def _value_or_default(self, obj, key, default):

NoSuchElementException: Message: no such element: Unable to locate element: {"method":"xpath","selector":"//*[@id=\"srp-body\"]/section[1]/div[2]/div/a[8]"}
  (Session info: chrome=92.0.4515.159)
  (Driver info: chromedriver=2.42.591059 (a3d9684d10d61aa0c45f6723b327283be1ebaad8),platform=Mac OS X 10.13.6 x86_64)
解决方案

最优方案:直接调用后端API获取数据

打开浏览器开发者工具的网络面板,筛选XHR请求,可找到realtor.com返回房源数据的后端接口,接口直接返回JSON格式的全量房源字段,无需解析HTML,请求时传入邮编、页码参数即可,效率是解析HTML的3-5倍,也不存在分页元素定位的问题。

次优方案:优化拼接URL的请求逻辑

你原来的分页URL规则是正确的,代码问题可通过以下修改解决:

  • 先从第一页提取总房源数,除以每页47条得到总页码,循环到总页码即可终止,不要用房源卡片是否为空作为终止条件,避免反爬返回空页面时提前终止
  • 增加响应状态码判断,若返回403/503等反爬状态码,可重试1-2次后再继续
  • 把等待时间缩短到10-20秒,请求头增加Referer、Accept字段降低反爬概率
  • 修正代码中len(soup.select('.component_property-card')[0])的错误写法,应该直接获取列表长度:len(soup.select('.component_property-card'))

Selenium方案问题修复

你使用的绝对路径XPath过于脆弱,页面结构稍有变化就会定位失败,可修改为相对路径定位Next按钮,同时增加显式等待:

  • 把定位XPath改为//a[@aria-label="Next Page"]或者//a[contains(text(),"Next")]
  • 导入selenium.webdriver.support.ui.WebDriverWait和selenium.webdriver.support.expected_conditions,等待按钮可点击后再执行点击操作,不要用固定时长的sleep
优化后的分页请求示例代码
from bs4 import BeautifulSoup
import requests
import time
import numpy as np

headers = {
    'User-Agent':'Mozilla/5.0 (Windows NT 6.1) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/88.0.4324.150 Safari/537.36',
    'Accept-Encoding': 'identity',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
    'Referer': 'https://www.realtor.com/'
}
base_url = 'https://www.realtor.com/realestateandhomes-search/94016'
all_soups = []

# 先请求第一页获取总页数
res = requests.get(base_url, headers=headers)
if res.status_code == 200:
    first_soup = BeautifulSoup(res.text, 'lxml')
    all_soups.append(first_soup)
    # 提取总房源数,根据页面实际元素调整选择器
    total_count = int(first_soup.select_one('[data-label=results-count]').text.replace(',', ''))
    total_page = (total_count // 47) + 1
    print(f"总房源数{total_count},总页数{total_page}")

    # 循环请求后续页
    for page in range(2, total_page+1):
        time.sleep(np.random.randint(10,20))
        page_url = f"{base_url}/pg-{page}"
        try:
            res = requests.get(page_url, headers=headers, timeout=15)
            if res.status_code == 200:
                soup = BeautifulSoup(res.text, 'lxml')
                all_soups.append(soup)
                print(f"第{page}页请求完成,房源数{len(soup.select('.component_property-card'))}")
            else:
                print(f"第{page}页请求失败,状态码{res.status_code}")
        except Exception as e:
            print(f"第{page}页请求异常:{str(e)}")

内容的提问来源于stack exchange,提问作者user3476463

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.10.07 06:54:03