You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

添加WebDriverWait后Selenium Python脚本触发TimeoutException求助

Selenium爬取折叠区域数据触发TimeoutException问题

问题背景

之前的Selenium Python代码爬取数据基本正常,但数据集存在异常——目标数据位于页面折叠区域下方。添加WebDriverWait代码点击箭头展开产品信息后,触发了TimeoutException,仅首屏上方的数据能正常爬取。

原代码

from selenium import webdriver
from selenium.common.exceptions import NoSuchElementException
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as EC
import pandas as pd

data = []

for y in range(1,3):
    website = f'https://www.knowde.com/b/markets-personal-care/products/{y}'
    path = '/Users/kdavid3mbp/Python/chrome_driver64/chromedriver'
    driver = webdriver.Chrome(path)
    driver.get(website)
    
    for x in range(1,37):
        products = driver.find_elements('xpath', f'//*[@id="__next"]/main/div/div[3]/div[3]/div[1]/div[2]/div[{x}]')

        for product in products:
            WebDriverWait(driver, 10).until(EC.element_to_be_clickable(('xpath', './div/div/svg'))).click()
            
            brand = product.find_element('xpath', './a/div[2]/div/p[1]').text
            item = product.find_element('xpath', './a/div[2]/div/p[2]').text
            inci_name = product.find_element('xpath', './a/div[2]/div/div[1]/span[2]').text
            try:
                ingredient_origin = product.find_element('xpath', './a/div[2]/div/div[3]/span[2]').text
            except NoSuchElementException:
                ingredient_origin = 'null'
            try:
                function = product.find_element('xpath', './a/div[2]/div/div[2]/span[2]').text
            except NoSuchElementException:
                function = 'null'
            try:
                benefit_claims = product.find_element('xpath', './a/div[2]/div/div[4]/span[2]').text
            except NoSuchElementException:
                benefit_claims = 'null'
            try:
                description = product.find_element('xpath', './a/div[2]/div/p[3]').text
            except NoSuchElementException:
                description = 'null'
            try:
                labeling_claims = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text
            except NoSuchElementException:
                labeling_claims = 'null'
            try:
                compliance = product.find_element('xpath', './a/div[2]/div/div[6]/span[2]').text
            except NoSuchElementException:
                compliance = 'null'
            try:
                hlb_value = product.find_element('xpath', './a/div[2]/div/div[4]/span[2]').text
            except NoSuchElementException:
                hlb_value = 'null'
            try:
                end_uses = product.find_element('xpath', '/a/div[2]/div/div[4]/span[2]').text
            except NoSuchElementException:
                end_uses = 'null'
            try:
                cas_no = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text
            except NoSuchElementException:
                cas_no = 'null'
            try:
                chemical_name = product.find_element('xpath', './a/div[2]/div/div[2]/span[2]').text
            except NoSuchElementException:
                chemical_name = 'null'
            try:
                synonyms = product.find_element('xpath', './a/div[2]/div/div[6]/span[2]').text
            except NoSuchElementException:
                synonyms = 'null'
            try:
                chemical_family = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text
            except NoSuchElementException:
                chemical_family = 'null'
            try:
                features = product.find_element('xpath', './a/div[2]/div/div[7]/span[2]').text
            except NoSuchElementException:
                features = 'null'
            try:
                grade = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text
            except NoSuchElementException:
                grade = 'null'

        dict = {
            'brand': brand,
            'item': item,
            'inci_name': inci_name,
            'ingredient_origin': ingredient_origin,
            'function': function,
            'benefit_claims': benefit_claims,
            'description': description,
            'labeling_claims': labeling_claims,
            'compliance': compliance,
            'hlb_value': hlb_value,
            'end_uses': end_uses,
            'cas_no': cas_no,
            'chemical_name': chemical_name,
            'synonyms': synonyms,
            'chemical_family': chemical_family,
            'features': features,
            'grade': grade
        }

        data.append(dict)
        print('Saving: ', dict['brand'])

# Closes driver once for loop is completed
driver.quit()

df = pd.DataFrame(data)
df.to_csv('/Users/kdavid3mbp/Python/cosmetics_data.csv', index=False)

报错信息

---------------------------------------------------------------------------
TimeoutException                          Traceback (most recent call last)
/var/folders/90/82_f843n4h9drvxh7z3tqg840000gn/T/ipykernel_34523/974946269.py in <module>
     18 
     19         for product in products:
---&gt; 20             WebDriverWait(product, 10).until(EC.element_to_be_clickable(('xpath', './div/div/svg'))).click()
     21 
     22             brand = product.find_element('xpath', './a/div[2]/div/p[1]').text

~/opt/anaconda3/lib/python3.9/site-packages/selenium/webdriver/support/wait.py in until(self, method, message)
     93             if time.monotonic() > end_time:
     94                 break
---&gt; 95         raise TimeoutException(message, screen, stacktrace)
     96 
     97     def until_not(self, method, message: str = ""):

TimeoutException: Message: 
Stacktrace:
0   chromedriver                        0x0000000106e946b8 chromedriver + 4937400
1   chromedriver                        0x0000000106e8bb73 chromedriver + 4901747
2   chromedriver                        0x0000000106a49616 chromedriver + 435734
3   chromedriver                        0x0000000106a8ce0f chromedriver + 712207
4   chromedriver                        0x0000000106a8d0a1 chromedriver + 712865
5   chromedriver                        0x0000000106a80ae6 chromedriver + 662246
6   chromedriver                        0x0000000106ab103d chromedriver + 860221
7   chromedriver                        0x0000000106a809c1 chromedriver + 661953
8   chromedriver                        0x0000000106ab11ce chromedriver + 860622
9   chromedriver                        0x0000000106acbe76 chromedriver + 970358
10  chromedriver                        0x0000000106ab0de3 chromedriver + 859619
11  chromedriver                        0x0000000106a7ed7f chromedriver + 654719
12  chromedriver                        0x0000000106a800de chromedriver + 659678
13  chromedriver                        0x0000000106e502ad chromedriver + 4657837
14  chromedriver                        0x0000000106e55130 chromedriver + 4677936
15  chromedriver                        0x0000000106e5bdef chromedriver + 4705775
16  chromedriver                        0x0000000106e5605a chromedriver + 4681818
17  chromedriver                        0x0000000106e2892c chromedriver + 4495660
18  chromedriver                        0x0000000106e73838 chromedriver + 4802616
19  chromedriver                        0x0000000106e739b7 chromedriver + 4802999
20  chromedriver                        0x0000000106e8499f chromedriver + 4872607
21  libsystem_pthread.dylib             0x00007ff81308d1d3 _pthread_start + 125
22  libsystem_pthread.dylib             0x00007ff813088bd3 thread_start + 15

问题分析与解决方案

核心问题

  1. WebDriverWait使用错误:不能将WebElement对象传入WebDriverWait,必须用driver实例;且./div/div/svg可能不是实际可点击的元素(通常svg是图标,点击区域是其父div)。
  2. 循环逻辑不稳定:通过索引div[{x}]定位产品元素,页面结构变化就会失效,应直接定位所有产品元素列表。
  3. 元素可见性问题:折叠元素可能在视口外,未滚动到对应位置就点击会导致无法触发。
  4. xpath重复错误:多个字段复用同一xpath,导致数据混乱。
  5. Driver资源浪费:每循环一次就创建新的Driver实例,影响性能。

修正后的代码

from selenium import webdriver
from selenium.common.exceptions import NoSuchElementException, TimeoutException
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.action_chains import ActionChains
import pandas as pd

data = []

# 初始化Driver,放在循环外复用
path = '/Users/kdavid3mbp/Python/chrome_driver64/chromedriver'
driver = webdriver.Chrome(path)
wait = WebDriverWait(driver, 10)

for y in range(1,3):
    website = f'https://www.knowde.com/b/markets-personal-care/products/{y}'
    driver.get(website)
    
    # 直接定位所有产品元素,无需索引遍历
    products = wait.until(EC.presence_of_all_elements_located((By.XPATH, '//*[@id="__next"]/main/div/div[3]/div[3]/div[1]/div[2]/div')))
    
    for product in products:
        try:
            # 滚动到产品元素位置,确保可见
            ActionChains(driver).move_to_element(product).perform()
            # 定位可点击的展开按钮(优先父div,而非svg)
            expand_btn = wait.until(EC.element_to_be_clickable((By.XPATH, './/div[contains(@class, "expand-icon-container")]')))
            # 检查是否已展开,避免重复点击
            if 'expanded' not in product.get_attribute('class'):
                expand_btn.click()
            # 等待展开后的数据加载完成
            wait.until(EC.presence_of_element_located((By.XPATH, './/a/div[2]/div/p[3]')))
        except TimeoutException:
            # 部分产品可能默认已展开,跳过点击
            pass
        
        # 提取数据,修正重复xpath问题(需根据实际页面结构调整字段路径)
        brand = product.find_element(By.XPATH, './/a/div[2]/div/p[1]').text
        item = product.find_element(By.XPATH, './/a/div[2]/div/p[2]').text
        inci_name = product.find_element(By.XPATH, './/a/div[2]/div/div[1]/span[2]').text
        
        # 简化元素存在性判断
        ingredient_origin = product.find_element(By.XPATH, './/a/div[2]/div/div[3]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[3]/span[2]') else 'null'
        function = product.find_element(By.XPATH, './/a/div[2]/div/div[2]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[2]/span[2]') else 'null'
        benefit_claims = product.find_element(By.XPATH, './/a/div[2]/div/div[4]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[4]/span[2]') else 'null'
        description = product.find_element(By.XPATH, './/a/div[2]/div/p[3]').text if product.find_elements(By.XPATH, './/a/div[2]/div/p[3]') else 'null'
        labeling_claims = product.find_element(By.XPATH, './/a/div[2]/div/div[5]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[5]/span[2]') else 'null'
        compliance = product.find_element(By.XPATH, './/a/div[2]/div/div[6]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[6]/span[2]') else 'null'
        
        # 修正各字段的唯一xpath(需对照页面真实结构调整)
        hlb_value = product.find_element(By.XPATH, './/a/div[2]/div/div[7]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[7]/span[2]') else 'null'
        end_uses = product.find_element(By.XPATH, './/a/div[2]/div/div[8]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[8]/span[2]') else 'null'
        cas_no = product.find_element(By.XPATH, './/a/div[2]/div/div[9]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[9]/span[2]') else 'null'
        chemical_name = product.find_element(By.XPATH, './/a/div[2]/div/div[10]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[10]/span[2]') else 'null'
        synonyms = product.find_element(By.XPATH, './/a/div[2]/div/div[11]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[11]/span[2]') else 'null'
        chemical_family = product.find_element(By.XPATH, './/a/div[2]/div/div[12]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[12]/span[2]') else 'null'
        features = product.find_element(By.XPATH, './/a/div[2]/div/div[13]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[13]/span[2]') else 'null'
        grade = product.find_element(By.XPATH, './/a/div[2]/div/div[14]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[14]/span[2]') else 'null'
        
        product_dict = {
            'brand': brand,
            'item': item,
            'inci_name': inci_name,
            'ingredient_origin': ingredient_origin,
            'function': function,
            'benefit_claims': benefit_claims,
            'description': description,
            'labeling_claims': labeling_claims,
            'compliance': compliance,
            'hlb_value': h
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.18 23:59:41