Selenium WebDriver如何等待所有动态加载的新元素加载完成?
问题描述
爬取IMDB电影上映日期页面(示例页面加载后仅显示5行,需点击"Show more"按钮获取完整数据),目标是下载包含所有上映日期的HTML。使用Selenium的流程如下:
- 访问URL,等待初始行元素加载完成
- 等待"Show more"按钮可点击后点击
- 等待所有新行加载完成
- 下载HTML
但第三步等待结果不稳定:有时HTML包含全部行,有时仅部分,有时还是初始5行。当前使用的等待条件和第一步相同:
all_rows = WebDriverWait(driver, waittime).until(EC.presence_of_all_elements_located((By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]')))
完整Python代码如下:
# library from selenium import webdriver from selenium.common.exceptions import NoSuchElementException from selenium.common.exceptions import TimeoutException from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC # def driver options def opening_default_driver(): chrome_options = webdriver.ChromeOptions() chrome_options.add_argument('--no-sandbox') chrome_options.add_argument("--disable-dev-shm-usage") chrome_options.add_argument('start-maximized') chrome_options.add_argument("--disable-infobars") chrome_options.add_argument("--disable-extensions") chrome_options.add_experimental_option("prefs", { \ "profile.default_content_setting_values.media_stream_mic": 2, "profile.default_content_setting_values.media_stream_camera": 2, "profile.default_content_setting_values.geolocation": 2, "profile.default_content_setting_values.notifications": 2 }) driver = webdriver.Chrome(options=chrome_options) return driver # download the html driver = opening_default_driver() waittime = 20 url = 'https://www.imdb.com/title/tt0929632/releaseinfo/' for t in range(0,20): try: driver.get(url) except Exception as e: print(f'timeoutexc {e} getting url: ind, {url}') else: all_rows = WebDriverWait(driver, waittime).until( EC.presence_of_all_elements_located((By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]')) ) try: button2_parent = driver.find_element(By.CSS_SELECTOR, '.ipc-see-more.sc-68fe39e1-0.icyVUF.chained-see-more-button-releases.sc-2e6342b6-1.gXymKs') button2 = button2_parent.find_element(By.TAG_NAME, 'button') except NoSuchElementException: try: button1_parent = driver.find_element(By.CSS_SELECTOR, '.ipc-see-more.sc-f06d8e21-0.jBTuow.single-page-see-more-button-releases') button1 = button1_parent.find_element(By.TAG_NAME, 'button') except NoSuchElementException: print(f'can\'t find neither button') else: button1.click() else: button2.click() finally: try: all_rows = WebDriverWait(driver, waittime).until( EC.presence_of_all_elements_located((By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]')) ) except TimeoutException: print(f'TimeoutException') else: f = f'/Users/lalala/Desktop/website{t}.html' html = driver.page_source with open(f, 'w+', encoding='utf-8') as file: file.write(html) driver.quit()
问题原因
presence_of_all_elements_located的局限性:这个条件只要页面中存在至少一个匹配元素就会返回,不会等待所有元素加载完成。点击"Show more"后IMDB是逐步渲染新行的,只要有新行出现,该条件就会触发,导致提前获取HTML,后续行还未加载。- 按钮定位不稳定:你使用了IMDB动态生成的CSS类名(如
.icyVUF、.jBTuow),这类类名会随页面更新变化,可能导致按钮点击失败,无法加载全部行。 - 未处理按钮状态变化:点击"Show more"后,按钮可能变为"Show less"或直接消失(所有内容加载完成时),你没有等待按钮状态变化就去等待行元素,逻辑存在间隙。
解决方案
1. 优化按钮定位逻辑
使用更稳定的定位方式,比如通过按钮文本或语义化属性:
# 定位包含"Show more"或"See more"文本的可点击按钮 show_more_button = WebDriverWait(driver, waittime).until( EC.element_to_be_clickable((By.XPATH, '//button[contains(text(), "Show more") or contains(text(), "See more")]')) ) show_more_button.click()
2. 自定义等待条件:等待行元素数量稳定
因为不知道最终行数量,我们可以等待连续一段时间内行元素数量不再变化,以此判断所有行加载完成:
def wait_for_elements_stable(driver, by, locator, timeout=20, interval=0.5): """等待元素数量稳定,连续2次检查数量不变则返回""" last_count = -1 stable_count = 0 end_time = time.time() + timeout while time.time() < end_time: current_elements = driver.find_elements(by, locator) current_count = len(current_elements) if current_count == last_count: stable_count += 1 if stable_count >= 2: return current_elements else: last_count = current_count stable_count = 0 time.sleep(interval) raise TimeoutException("元素数量在超时时间内未稳定")
3. 完整修改后的代码
# library import time from selenium import webdriver from selenium.common.exceptions import NoSuchElementException from selenium.common.exceptions import TimeoutException from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC # def driver options def opening_default_driver(): chrome_options = webdriver.ChromeOptions() chrome_options.add_argument('--no-sandbox') chrome_options.add_argument("--disable-dev-shm-usage") chrome_options.add_argument('start-maximized') chrome_options.add_argument("--disable-infobars") chrome_options.add_argument("--disable-extensions") chrome_options.add_experimental_option("prefs", { "profile.default_content_setting_values.media_stream_mic": 2, "profile.default_content_setting_values.media_stream_camera": 2, "profile.default_content_setting_values.geolocation": 2, "profile.default_content_setting_values.notifications": 2 }) driver = webdriver.Chrome(options=chrome_options) return driver def wait_for_elements_stable(driver, by, locator, timeout=20, interval=0.5): last_count = -1 stable_count = 0 end_time = time.time() + timeout while time.time() < end_time: current_elements = driver.find_elements(by, locator) current_count = len(current_elements) if current_count == last_count: stable_count += 1 if stable_count >= 2: return current_elements else: last_count = current_count stable_count = 0 time.sleep(interval) raise TimeoutException("元素数量在超时时间内未稳定") # download the html driver = opening_default_driver() waittime = 20 url = 'https://www.imdb.com/title/tt0929632/releaseinfo/' row_locator = (By.XPATH, '//div[@data-testid="sub-section-releases"]//li[@data-testid="list-item"]') for t in range(0,20): try: driver.get(url) except Exception as e: print(f'timeoutexc {e} getting url: ind, {url}') continue # 等待初始行加载完成 wait_for_elements_stable(driver, *row_locator, timeout=waittime) # 尝试点击Show more按钮 try: show_more_button = WebDriverWait(driver, waittime).until( EC.element_to_be_clickable((By.XPATH, '//button[contains(text(), "Show more") or contains(text(), "See more")]')) ) show_more_button.click() # 等待按钮状态变化(变为Show less或从DOM中移除) WebDriverWait(driver, waittime).until( lambda d: EC.staleness_of(show_more_button)(d) or "Show less" in d.find_element(By.XPATH, '//button[contains(text(), "Show less")]').text ) except (NoSuchElementException, TimeoutException): print(f'No show more button or failed to click') # 等待所有行加载完成(数量稳定) try: wait_for_elements_stable(driver, *row_locator, timeout=waittime) except TimeoutException: print(f'Timeout waiting for rows to stabilize') else: f = f'/Users/lalala/Desktop/website{t}.html' html = driver.page_source with open(f, 'w+', encoding='utf-8') as file: file.write(html) driver.quit()
额外优化点
- 增加
continue跳过异常迭代,避免无效执行 - 用
staleness_of判断按钮是否失效,确保点击操作完成 - 提取行定位器为变量,减少重复代码
内容的提问来源于stack exchange,提问作者iim7b5-v7-im7
相关产品推荐
相关产品推荐

