You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python Playwright处理Booking.com分页爬取问题

解决Playwright爬取Booking.com多页搜索结果的分页问题

方法一:模拟点击分页按钮(贴近用户行为)

Booking.com的分页栏提供「Next page」按钮,我们可以循环点击该按钮直到按钮不可用(到达最后一页),核心是处理页面加载与按钮状态判断。

修改后的完整代码:

from playwright.sync_api import sync_playwright, expect
import pandas as pd

def main():
    hotels_list = []
    with sync_playwright() as p:
        checkin_date = '2023-12-10'
        checkout_date = '2023-12-18'
        base_url = f'https://www.booking.com/searchresults.html?ss=Medina%2C+Saudi+Arabia&label=gog235jc-1DCAEoggI46AdIM1gDaMQBiAEBmAExuAEXyAEP2AED6AEB-AECiAIBqAIDuAL4_M2qBsACAdICJDIyNmY5NThlLTdkNjctNDg2Yi05ZDMzLWY3M2JhZmRkZDdhNtgCBOACAQ&aid=397594&lang=en-us&sb=1&src_elem=sb&src=index&dest_id=-3092186&dest_type=city&checkin={checkin_date}&checkout={checkout_date}&group_adults=1&no_rooms=1&group_children=0&sb_travel_purpose=leisure'
        
        browser = p.chromium.launch(headless=False)
        page = browser.new_page()
        page.goto(base_url, timeout=60000)
        
        while True:
            # 等待酒店卡片加载完成
            page.wait_for_selector('//div[@data-testid="property-card"]', timeout=30000)
            
            # 爬取当前页数据
            hotels = page.locator('//div[@data-testid="property-card"]').all()
            print(f'当前页爬取到 {len(hotels)} 家酒店')
            
            for hotel in hotels:
                hotel_dict = {}
                try:
                    hotel_dict['hotel'] = hotel.locator('//div[@data-testid="title"]').inner_text()
                    hotel_dict['price'] = hotel.locator('//span[@data-testid="price-and-discounted-price"]').inner_text()
                    hotel_dict['Nights'] = hotel.locator('//div[@data-testid="availability-rate-wrapper"]/div[1]/div[1]').inner_text()
                    hotel_dict['tax'] = hotel.locator('//div[@data-testid="availability-rate-wrapper"]/div[1]/div[3]').inner_text()
                    hotel_dict['score'] = hotel.locator('//div[@data-testid="review-score"]/div[1]').inner_text()
                    hotel_dict['distance'] = hotel.locator('//span[@data-testid="distance"]').inner_text()
                    hotel_dict['avg review'] = hotel.locator('//div[@data-testid="review-score"]/div[2]/div[1]').inner_text()
                    hotel_dict['reviews count'] = hotel.locator('//div[@data-testid="review-score"]/div[2]/div[2]').inner_text().split()[0]
                except Exception as e:
                    print(f"爬取酒店信息出错: {e}")
                    continue
                hotels_list.append(hotel_dict)
            
            # 尝试点击下一页
            try:
                next_btn = page.locator('//button[@data-testid="pagination-next-arrow"]')
                expect(next_btn).not_to_be_disabled()
                next_btn.click()
                page.wait_for_url(f"{base_url}&page=*", timeout=30000)
            except:
                print("已爬取到最后一页")
                break
        
        # 保存数据
        df = pd.DataFrame(hotels_list)
        df.to_excel('hotels_list.xlsx', index=False) 
        df.to_csv('hotels_list.csv', index=False) 
        browser.close()
        
if __name__ == '__main__':
    main()

核心细节:

  • 用page.wait_for_selector确保酒店卡片加载完成后再爬取
  • 通过expect判断下一页按钮是否可用,避免无效点击
  • 加入异常捕获,防止单个酒店字段缺失导致程序崩溃

方法二:通过URL参数控制页码(效率更高)

Booking.com的搜索URL支持page参数(如&page=2表示第二页),直接构造不同页码的URL循环请求,跳过点击操作。

修改后的完整代码:

from playwright.sync_api import sync_playwright
import pandas as pd

def main():
    hotels_list = []
    with sync_playwright() as p:
        checkin_date = '2023-12-10'
        checkout_date = '2023-12-18'
        base_url = f'https://www.booking.com/searchresults.html?ss=Medina%2C+Saudi+Arabia&label=gog235jc-1DCAEoggI46AdIM1gDaMQBiAEBmAExuAEXyAEP2AED6AEB-AECiAIBqAIDuAL4_M2qBsACAdICJDIyNmY5NThlLTdkNjctNDg2Yi05ZDMzLWY3M2JhZmRkZDdhNtgCBOACAQ&aid=397594&lang=en-us&sb=1&src_elem=sb&src=index&dest_id=-3092186&dest_type=city&checkin={checkin_date}&checkout={checkout_date}&group_adults=1&no_rooms=1&group_children=0&sb_travel_purpose=leisure'
        
        browser = p.chromium.launch(headless=False)
        page = browser.new_page()
        
        # 可先爬第一页获取总页数,再动态设置max_pages
        max_pages = 5 
        for page_num in range(1, max_pages + 1):
            page_url = f"{base_url}&page={page_num}"
            page.goto(page_url, timeout=60000)
            
            page.wait_for_selector('//div[@data-testid="property-card"]', timeout=30000)
            
            hotels = page.locator('//div[@data-testid="property-card"]').all()
            print(f'第 {page_num} 页爬取到 {len(hotels)} 家酒店')
            
            for hotel in hotels:
                hotel_dict = {}
                try:
                    hotel_dict['hotel'] = hotel.locator('//div[@data-testid="title"]').inner_text()
                    hotel_dict['price'] = hotel.locator('//span[@data-testid="price-and-discounted-price"]').inner_text()
                    hotel_dict['Nights'] = hotel.locator('//div[@data-testid="availability-rate-wrapper"]/div[1]/div[1]').inner_text()
                    hotel_dict['tax'] = hotel.locator('//div[@data-testid="availability-rate-wrapper"]/div[1]/div[3]').inner_text()
                    hotel_dict['score'] = hotel.locator('//div[@data-testid="review-score"]/div[1]').inner_text()
                    hotel_dict['distance'] = hotel.locator('//span[@data-testid="distance"]').inner_text()
                    hotel_dict['avg review'] = hotel.locator('//div[@data-testid="review-score"]/div[2]/div[1]').inner_text()
                    hotel_dict['reviews count'] = hotel.locator('//div[@data-testid="review-score"]/div[2]/div[2]').inner_text().split()[0]
                except Exception as e:
                    print(f"爬取酒店信息出错: {e}")
                    continue
                hotels_list.append(hotel_dict)
        
        # 保存数据
        df = pd.DataFrame(hotels_list)
        df.to_excel('hotels_list.xlsx', index=False) 
        df.to_csv('hotels_list.csv', index=False) 
        browser.close()
        
if __name__ == '__main__':
    main()

核心细节:

  • 直接构造带页码参数的URL,减少页面交互步骤,提升效率
  • 可通过第一页的分页栏元素(如//div[@data-testid="pagination-page-count"])获取总页数,动态设置爬取范围

通用注意事项

  • 反爬应对:添加随机等待时间(如page.wait_for_timeout(random.randint(1500, 3000))),避免请求频率过高
  • 元素维护:Booking.com的data-testid可能会更新,需定期验证定位器有效性
  • 数据完整性:对可能缺失的字段(如无评分的酒店),设置默认值或跳过处理

内容的提问来源于stack exchange,提问作者Ahmed

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.06 11:37:05