代码可运行但返回空DataFrame:OpenTable爬虫故障求助
问题:OpenTable爬取代码返回空DataFrame
几年前编写的OpenTable数据爬取代码可正常运行,但返回空DataFrame。原代码及运行结果如下:
原代码
from selenium import webdriver import pandas as pd from bs4 import BeautifulSoup from time import sleep import re def parse_html(html): data, item = pd.DataFrame(), {} soup = BeautifulSoup(html, 'lxml') for i, resto in enumerate(soup.find_all('div', class_='rest-row-info')): item['name'] = resto.find('span', class_='rest-row-name-text').text booking = resto.find('div', class_='booking') item['bookings'] = re.search('\\d+', booking.text).group() if booking else 'NA' rating = resto.select('.star-rating .star-rating-score') #print(rating) item['rating'] = rating[0]['aria-label'] if rating else 'NA' reviews = resto.find('span', class_='star-rating-text--review-text') reviews = resto.select('div.review-rating-text span') print(reviews) item['reviews'] = reviews[0].text if reviews else 'NA' item['price'] = int(resto.find('div', class_='rest-row-pricing').find('i').text.count('$')) item['cuisine'] = resto.find_all('span', class_='rest-row-meta--cuisine')[-1].text #print(item['cuisine']) item['location'] = resto.find('span', class_='rest-row-meta--location').text data[i] = pd.Series(item) return data.T restaurants = pd.DataFrame() #driver = webdriver.Chrome(ChromeDriverManager().install()) driver = webdriver.Chrome() url = "https://www.opentable.com/s?dateTime=2022-11-15T19%3A00%3A00&covers=2&metroId=21&regionIds%5B0%5D=251&neighborhoodIds%5B0%5D=&term=&originCorrelationId=e1dada45-cc11-4711-848f-825e79b3ef30" driver.get(url) while True: sleep(1) new_data = parse_html(driver.page_source) if new_data.empty: break restaurants = pd.concat([restaurants, new_data], ignore_index=True) print(len(restaurants)) # driver.find_element_by_link_text('Next').click() #driver.close() restaurants.to_csv('results.csv', index=False) print(restaurants)
运行结果
Empty DataFrame Columns: [] Index: []
排查原因
- 页面结构变更:OpenTable的页面元素类名(如
rest-row-info)已更新,原代码依赖的选择器全部失效,导致BeautifulSoup无法找到餐厅元素。 - 等待机制不足:固定
sleep(1)无法保证页面动态加载完成,提前解析会获取空页面内容。 - 容错逻辑缺失:部分字段未做空值判断,若单个元素不存在可能导致解析中断。
修复方案
修复后的代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pandas as pd from bs4 import BeautifulSoup from time import sleep import re def parse_html(html): data = [] soup = BeautifulSoup(html, 'lxml') # 使用当前页面的餐厅卡片类名 for resto in soup.find_all('div', class_='resto-card'): item = {} # 餐厅名称 name_elem = resto.find('h3', class_='resto-name') item['name'] = name_elem.text.strip() if name_elem else 'NA' # 预订人数 booking_elem = resto.find('div', class_='booking-cta') if booking_elem: match = re.search(r'\d+', booking_elem.text) item['bookings'] = match.group() if match else 'NA' else: item['bookings'] = 'NA' # 评分 rating_elem = resto.find('span', class_='sr-only') item['rating'] = rating_elem.text.strip() if rating_elem else 'NA' # 评论数 reviews_elem = resto.find('span', class_='review-count') if reviews_elem: match = re.search(r'\d+', reviews_elem.text) item['reviews'] = match.group() if match else 'NA' else: item['reviews'] = 'NA' # 价格等级 price_elem = resto.find('div', class_='price-range') item['price'] = price_elem.text.count('$') if price_elem else 0 # 菜系 cuisine_elem = resto.find('div', class_='cuisine') item['cuisine'] = cuisine_elem.text.strip() if cuisine_elem else 'NA' # 位置 location_elem = resto.find('div', class_='location') item['location'] = location_elem.text.strip() if location_elem else 'NA' data.append(item) return pd.DataFrame(data) restaurants = pd.DataFrame() driver = webdriver.Chrome() url = "https://www.opentable.com/s?dateTime=2022-11-15T19%3A00%3A00&covers=2&metroId=21®ionIds%5B0%5D=251&neighborhoodIds%5B0%5D=&term=&originCorrelationId=e1dada45-cc11-4711-848f-825e79b3ef30" driver.get(url) wait = WebDriverWait(driver, 10) while True: # 等待餐厅卡片加载完成 try: wait.until(EC.presence_of_element_located((By.CLASS_NAME, 'resto-card'))) except: break new_data = parse_html(driver.page_source) if new_data.empty: break restaurants = pd.concat([restaurants, new_data], ignore_index=True) print(f"已爬取 {len(restaurants)} 条数据") # 点击下一页,处理无下一页的情况 try: next_btn = driver.find_element(By.CSS_SELECTOR, 'button[data-testid="pagination-next"]') if 'disabled' in next_btn.get_attribute('class'): break next_btn.click() sleep(2) except: break driver.close() restaurants.to_csv('results.csv', index=False) print(restaurants)
关键改动说明
- 更新元素选择器:全部替换为OpenTable当前页面使用的类名,可通过浏览器开发者工具(F12)实时查看元素结构。
- 增强等待逻辑:使用
WebDriverWait替代固定休眠,确保页面元素加载完成后再解析。 - 优化容错处理:每个字段都添加空值判断,避免因单个元素缺失导致代码崩溃。
- 修复分页功能:更新下一页按钮选择器,并判断按钮是否禁用,避免无效点击或无限循环。
内容的提问来源于stack exchange,提问作者user16128779
相关产品推荐
相关产品推荐

