You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

代码可运行但返回空DataFrame:OpenTable爬虫故障求助

问题:OpenTable爬取代码返回空DataFrame

几年前编写的OpenTable数据爬取代码可正常运行,但返回空DataFrame。原代码及运行结果如下:

原代码

from selenium import webdriver
import pandas as pd
from bs4 import BeautifulSoup
from time import sleep
import re

def parse_html(html):
    data, item = pd.DataFrame(), {}
    soup = BeautifulSoup(html, 'lxml')
    for i, resto in enumerate(soup.find_all('div', class_='rest-row-info')):
        item['name'] = resto.find('span', class_='rest-row-name-text').text

        booking = resto.find('div', class_='booking')
        item['bookings'] = re.search('\\d+', booking.text).group() if booking else 'NA'

        rating = resto.select('.star-rating .star-rating-score')
        #print(rating)
        item['rating'] = rating[0]['aria-label'] if rating else 'NA'

        reviews = resto.find('span', class_='star-rating-text--review-text')
        
        reviews = resto.select('div.review-rating-text span')
        print(reviews)
        item['reviews'] = reviews[0].text if reviews else 'NA'

        item['price'] = int(resto.find('div', class_='rest-row-pricing').find('i').text.count('$'))
        
        item['cuisine'] = resto.find_all('span', class_='rest-row-meta--cuisine')[-1].text
        #print(item['cuisine'])
        
        item['location'] = resto.find('span', class_='rest-row-meta--location').text
        data[i] = pd.Series(item)
    return data.T


restaurants = pd.DataFrame()
#driver = webdriver.Chrome(ChromeDriverManager().install())
driver = webdriver.Chrome()
url = "https://www.opentable.com/s?dateTime=2022-11-15T19%3A00%3A00&covers=2&metroId=21&regionIds%5B0%5D=251&neighborhoodIds%5B0%5D=&term=&originCorrelationId=e1dada45-cc11-4711-848f-825e79b3ef30"
driver.get(url)

while True:
    sleep(1)
    new_data = parse_html(driver.page_source)
    if new_data.empty:
        break
    restaurants = pd.concat([restaurants, new_data], ignore_index=True)
    print(len(restaurants))
   # driver.find_element_by_link_text('Next').click()
    
#driver.close()
restaurants.to_csv('results.csv', index=False)
print(restaurants)

运行结果

Empty DataFrame
Columns: []
Index: []

排查原因

  • 页面结构变更:OpenTable的页面元素类名(如rest-row-info)已更新,原代码依赖的选择器全部失效,导致BeautifulSoup无法找到餐厅元素。
  • 等待机制不足:固定sleep(1)无法保证页面动态加载完成,提前解析会获取空页面内容。
  • 容错逻辑缺失:部分字段未做空值判断,若单个元素不存在可能导致解析中断。

修复方案

修复后的代码

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import pandas as pd
from bs4 import BeautifulSoup
from time import sleep
import re

def parse_html(html):
    data = []
    soup = BeautifulSoup(html, 'lxml')
    # 使用当前页面的餐厅卡片类名
    for resto in soup.find_all('div', class_='resto-card'):
        item = {}
        # 餐厅名称
        name_elem = resto.find('h3', class_='resto-name')
        item['name'] = name_elem.text.strip() if name_elem else 'NA'
        
        # 预订人数
        booking_elem = resto.find('div', class_='booking-cta')
        if booking_elem:
            match = re.search(r'\d+', booking_elem.text)
            item['bookings'] = match.group() if match else 'NA'
        else:
            item['bookings'] = 'NA'
        
        # 评分
        rating_elem = resto.find('span', class_='sr-only')
        item['rating'] = rating_elem.text.strip() if rating_elem else 'NA'
        
        # 评论数
        reviews_elem = resto.find('span', class_='review-count')
        if reviews_elem:
            match = re.search(r'\d+', reviews_elem.text)
            item['reviews'] = match.group() if match else 'NA'
        else:
            item['reviews'] = 'NA'
        
        # 价格等级
        price_elem = resto.find('div', class_='price-range')
        item['price'] = price_elem.text.count('$') if price_elem else 0
        
        # 菜系
        cuisine_elem = resto.find('div', class_='cuisine')
        item['cuisine'] = cuisine_elem.text.strip() if cuisine_elem else 'NA'
        
        # 位置
        location_elem = resto.find('div', class_='location')
        item['location'] = location_elem.text.strip() if location_elem else 'NA'
        
        data.append(item)
    return pd.DataFrame(data)


restaurants = pd.DataFrame()
driver = webdriver.Chrome()
url = "https://www.opentable.com/s?dateTime=2022-11-15T19%3A00%3A00&covers=2&metroId=21&regionIds%5B0%5D=251&neighborhoodIds%5B0%5D=&term=&originCorrelationId=e1dada45-cc11-4711-848f-825e79b3ef30"
driver.get(url)

wait = WebDriverWait(driver, 10)
while True:
    # 等待餐厅卡片加载完成
    try:
        wait.until(EC.presence_of_element_located((By.CLASS_NAME, 'resto-card')))
    except:
        break
    
    new_data = parse_html(driver.page_source)
    if new_data.empty:
        break
    restaurants = pd.concat([restaurants, new_data], ignore_index=True)
    print(f"已爬取 {len(restaurants)} 条数据")
    
    # 点击下一页,处理无下一页的情况
    try:
        next_btn = driver.find_element(By.CSS_SELECTOR, 'button[data-testid="pagination-next"]')
        if 'disabled' in next_btn.get_attribute('class'):
            break
        next_btn.click()
        sleep(2)
    except:
        break

driver.close()
restaurants.to_csv('results.csv', index=False)
print(restaurants)

关键改动说明

  • 更新元素选择器:全部替换为OpenTable当前页面使用的类名,可通过浏览器开发者工具(F12)实时查看元素结构。
  • 增强等待逻辑:使用WebDriverWait替代固定休眠,确保页面元素加载完成后再解析。
  • 优化容错处理:每个字段都添加空值判断,避免因单个元素缺失导致代码崩溃。
  • 修复分页功能:更新下一页按钮选择器,并判断按钮是否禁用,避免无效点击或无限循环。

内容的提问来源于stack exchange,提问作者user16128779

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.12 18:30:54