You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium WebDriver如何等待所有动态加载的新元素加载完成?

问题描述

爬取IMDB电影上映日期页面(示例页面加载后仅显示5行,需点击"Show more"按钮获取完整数据),目标是下载包含所有上映日期的HTML。使用Selenium的流程如下:

  • 访问URL,等待初始行元素加载完成
  • 等待"Show more"按钮可点击后点击
  • 等待所有新行加载完成
  • 下载HTML

但第三步等待结果不稳定:有时HTML包含全部行,有时仅部分,有时还是初始5行。当前使用的等待条件和第一步相同:

all_rows = WebDriverWait(driver, waittime).until(EC.presence_of_all_elements_located((By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]')))

完整Python代码如下:

# library
from selenium import webdriver
from selenium.common.exceptions import NoSuchElementException
from selenium.common.exceptions import TimeoutException
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

# def driver options
def opening_default_driver():
  chrome_options = webdriver.ChromeOptions()
  chrome_options.add_argument('--no-sandbox') 
  chrome_options.add_argument("--disable-dev-shm-usage")
  chrome_options.add_argument('start-maximized')
  chrome_options.add_argument("--disable-infobars")
  chrome_options.add_argument("--disable-extensions")
  chrome_options.add_experimental_option("prefs", { \
      "profile.default_content_setting_values.media_stream_mic": 2, 
      "profile.default_content_setting_values.media_stream_camera": 2,
      "profile.default_content_setting_values.geolocation": 2, 
      "profile.default_content_setting_values.notifications": 2 
    })
  driver = webdriver.Chrome(options=chrome_options)
  return driver

# download the html
driver = opening_default_driver()
waittime = 20
url = 'https://www.imdb.com/title/tt0929632/releaseinfo/'

for t in range(0,20):
    try:
        driver.get(url)
    except Exception as e:
        print(f'timeoutexc {e} getting url: ind, {url}')
    else:
        all_rows = WebDriverWait(driver, waittime).until(
            EC.presence_of_all_elements_located((By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]'))
        )
        try:
            button2_parent = driver.find_element(By.CSS_SELECTOR, '.ipc-see-more.sc-68fe39e1-0.icyVUF.chained-see-more-button-releases.sc-2e6342b6-1.gXymKs')
            button2 = button2_parent.find_element(By.TAG_NAME, 'button')
        except NoSuchElementException:
            try:
                button1_parent = driver.find_element(By.CSS_SELECTOR, '.ipc-see-more.sc-f06d8e21-0.jBTuow.single-page-see-more-button-releases')
                button1 = button1_parent.find_element(By.TAG_NAME, 'button')
            except NoSuchElementException:
                print(f'can\'t find neither button')
            else:
                button1.click()
        else:
            button2.click()
        finally:
            try:
                all_rows = WebDriverWait(driver, waittime).until(
                    EC.presence_of_all_elements_located((By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]'))
                )
            except TimeoutException:
                print(f'TimeoutException')
            else:
                f = f'/Users/lalala/Desktop/website{t}.html'
                html = driver.page_source
                with open(f, 'w+', encoding='utf-8') as file:
                    file.write(html)

driver.quit()
问题原因
  1. presence_of_all_elements_located的局限性:这个条件只要页面中存在至少一个匹配元素就会返回,不会等待所有元素加载完成。点击"Show more"后IMDB是逐步渲染新行的,只要有新行出现,该条件就会触发,导致提前获取HTML,后续行还未加载。
  2. 按钮定位不稳定:你使用了IMDB动态生成的CSS类名(如.icyVUF、.jBTuow),这类类名会随页面更新变化,可能导致按钮点击失败,无法加载全部行。
  3. 未处理按钮状态变化:点击"Show more"后,按钮可能变为"Show less"或直接消失(所有内容加载完成时),你没有等待按钮状态变化就去等待行元素,逻辑存在间隙。
解决方案

1. 优化按钮定位逻辑

使用更稳定的定位方式,比如通过按钮文本或语义化属性:

# 定位包含"Show more"或"See more"文本的可点击按钮
show_more_button = WebDriverWait(driver, waittime).until(
    EC.element_to_be_clickable((By.XPATH, '//button[contains(text(), "Show more") or contains(text(), "See more")]'))
)
show_more_button.click()

2. 自定义等待条件:等待行元素数量稳定

因为不知道最终行数量,我们可以等待连续一段时间内行元素数量不再变化,以此判断所有行加载完成:

def wait_for_elements_stable(driver, by, locator, timeout=20, interval=0.5):
    """等待元素数量稳定,连续2次检查数量不变则返回"""
    last_count = -1
    stable_count = 0
    end_time = time.time() + timeout
    while time.time() < end_time:
        current_elements = driver.find_elements(by, locator)
        current_count = len(current_elements)
        if current_count == last_count:
            stable_count += 1
            if stable_count >= 2:
                return current_elements
        else:
            last_count = current_count
            stable_count = 0
        time.sleep(interval)
    raise TimeoutException("元素数量在超时时间内未稳定")

3. 完整修改后的代码

# library
import time
from selenium import webdriver
from selenium.common.exceptions import NoSuchElementException
from selenium.common.exceptions import TimeoutException
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

# def driver options
def opening_default_driver():
    chrome_options = webdriver.ChromeOptions()
    chrome_options.add_argument('--no-sandbox') 
    chrome_options.add_argument("--disable-dev-shm-usage")
    chrome_options.add_argument('start-maximized')
    chrome_options.add_argument("--disable-infobars")
    chrome_options.add_argument("--disable-extensions")
    chrome_options.add_experimental_option("prefs", { 
        "profile.default_content_setting_values.media_stream_mic": 2, 
        "profile.default_content_setting_values.media_stream_camera": 2,
        "profile.default_content_setting_values.geolocation": 2, 
        "profile.default_content_setting_values.notifications": 2 
    })
    driver = webdriver.Chrome(options=chrome_options)
    return driver

def wait_for_elements_stable(driver, by, locator, timeout=20, interval=0.5):
    last_count = -1
    stable_count = 0
    end_time = time.time() + timeout
    while time.time() < end_time:
        current_elements = driver.find_elements(by, locator)
        current_count = len(current_elements)
        if current_count == last_count:
            stable_count += 1
            if stable_count >= 2:
                return current_elements
        else:
            last_count = current_count
            stable_count = 0
        time.sleep(interval)
    raise TimeoutException("元素数量在超时时间内未稳定")

# download the html
driver = opening_default_driver()
waittime = 20
url = 'https://www.imdb.com/title/tt0929632/releaseinfo/'
row_locator = (By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]')

for t in range(0,20):
    try:
        driver.get(url)
    except Exception as e:
        print(f'timeoutexc {e} getting url: ind, {url}')
        continue
    
    # 等待初始行加载完成
    wait_for_elements_stable(driver, *row_locator, timeout=waittime)
    
    # 尝试点击Show more按钮
    try:
        show_more_button = WebDriverWait(driver, waittime).until(
            EC.element_to_be_clickable((By.XPATH, '//button[contains(text(), "Show more") or contains(text(), "See more")]'))
        )
        show_more_button.click()
        # 等待按钮状态变化(变为Show less或从DOM中移除)
        WebDriverWait(driver, waittime).until(
            lambda d: EC.staleness_of(show_more_button)(d) or "Show less" in d.find_element(By.XPATH, '//button[contains(text(), "Show less")]').text
        )
    except (NoSuchElementException, TimeoutException):
        print(f'No show more button or failed to click')
    
    # 等待所有行加载完成(数量稳定)
    try:
        wait_for_elements_stable(driver, *row_locator, timeout=waittime)
    except TimeoutException:
        print(f'Timeout waiting for rows to stabilize')
    else:
        f = f'/Users/lalala/Desktop/website{t}.html'
        html = driver.page_source
        with open(f, 'w+', encoding='utf-8') as file:
            file.write(html)

driver.quit()

额外优化点

  • 增加continue跳过异常迭代,避免无效执行
  • 用staleness_of判断按钮是否失效,确保点击操作完成
  • 提取行定位器为变量,减少重复代码

内容的提问来源于stack exchange,提问作者iim7b5-v7-im7

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.05 03:18:11