添加WebDriverWait后Selenium Python脚本触发TimeoutException求助
Selenium爬取折叠区域数据触发TimeoutException问题
问题背景
之前的Selenium Python代码爬取数据基本正常,但数据集存在异常——目标数据位于页面折叠区域下方。添加WebDriverWait代码点击箭头展开产品信息后,触发了TimeoutException,仅首屏上方的数据能正常爬取。
原代码
from selenium import webdriver from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.common.by import By from selenium.webdriver.support import expected_conditions as EC import pandas as pd data = [] for y in range(1,3): website = f'https://www.knowde.com/b/markets-personal-care/products/{y}' path = '/Users/kdavid3mbp/Python/chrome_driver64/chromedriver' driver = webdriver.Chrome(path) driver.get(website) for x in range(1,37): products = driver.find_elements('xpath', f'//*[@id="__next"]/main/div/div[3]/div[3]/div[1]/div[2]/div[{x}]') for product in products: WebDriverWait(driver, 10).until(EC.element_to_be_clickable(('xpath', './div/div/svg'))).click() brand = product.find_element('xpath', './a/div[2]/div/p[1]').text item = product.find_element('xpath', './a/div[2]/div/p[2]').text inci_name = product.find_element('xpath', './a/div[2]/div/div[1]/span[2]').text try: ingredient_origin = product.find_element('xpath', './a/div[2]/div/div[3]/span[2]').text except NoSuchElementException: ingredient_origin = 'null' try: function = product.find_element('xpath', './a/div[2]/div/div[2]/span[2]').text except NoSuchElementException: function = 'null' try: benefit_claims = product.find_element('xpath', './a/div[2]/div/div[4]/span[2]').text except NoSuchElementException: benefit_claims = 'null' try: description = product.find_element('xpath', './a/div[2]/div/p[3]').text except NoSuchElementException: description = 'null' try: labeling_claims = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text except NoSuchElementException: labeling_claims = 'null' try: compliance = product.find_element('xpath', './a/div[2]/div/div[6]/span[2]').text except NoSuchElementException: compliance = 'null' try: hlb_value = product.find_element('xpath', './a/div[2]/div/div[4]/span[2]').text except NoSuchElementException: hlb_value = 'null' try: end_uses = product.find_element('xpath', '/a/div[2]/div/div[4]/span[2]').text except NoSuchElementException: end_uses = 'null' try: cas_no = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text except NoSuchElementException: cas_no = 'null' try: chemical_name = product.find_element('xpath', './a/div[2]/div/div[2]/span[2]').text except NoSuchElementException: chemical_name = 'null' try: synonyms = product.find_element('xpath', './a/div[2]/div/div[6]/span[2]').text except NoSuchElementException: synonyms = 'null' try: chemical_family = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text except NoSuchElementException: chemical_family = 'null' try: features = product.find_element('xpath', './a/div[2]/div/div[7]/span[2]').text except NoSuchElementException: features = 'null' try: grade = product.find_element('xpath', './a/div[2]/div/div[5]/span[2]').text except NoSuchElementException: grade = 'null' dict = { 'brand': brand, 'item': item, 'inci_name': inci_name, 'ingredient_origin': ingredient_origin, 'function': function, 'benefit_claims': benefit_claims, 'description': description, 'labeling_claims': labeling_claims, 'compliance': compliance, 'hlb_value': hlb_value, 'end_uses': end_uses, 'cas_no': cas_no, 'chemical_name': chemical_name, 'synonyms': synonyms, 'chemical_family': chemical_family, 'features': features, 'grade': grade } data.append(dict) print('Saving: ', dict['brand']) # Closes driver once for loop is completed driver.quit() df = pd.DataFrame(data) df.to_csv('/Users/kdavid3mbp/Python/cosmetics_data.csv', index=False)
报错信息
--------------------------------------------------------------------------- TimeoutException Traceback (most recent call last) /var/folders/90/82_f843n4h9drvxh7z3tqg840000gn/T/ipykernel_34523/974946269.py in <module> 18 19 for product in products: ---> 20 WebDriverWait(product, 10).until(EC.element_to_be_clickable(('xpath', './div/div/svg'))).click() 21 22 brand = product.find_element('xpath', './a/div[2]/div/p[1]').text ~/opt/anaconda3/lib/python3.9/site-packages/selenium/webdriver/support/wait.py in until(self, method, message) 93 if time.monotonic() > end_time: 94 break ---> 95 raise TimeoutException(message, screen, stacktrace) 96 97 def until_not(self, method, message: str = ""): TimeoutException: Message: Stacktrace: 0 chromedriver 0x0000000106e946b8 chromedriver + 4937400 1 chromedriver 0x0000000106e8bb73 chromedriver + 4901747 2 chromedriver 0x0000000106a49616 chromedriver + 435734 3 chromedriver 0x0000000106a8ce0f chromedriver + 712207 4 chromedriver 0x0000000106a8d0a1 chromedriver + 712865 5 chromedriver 0x0000000106a80ae6 chromedriver + 662246 6 chromedriver 0x0000000106ab103d chromedriver + 860221 7 chromedriver 0x0000000106a809c1 chromedriver + 661953 8 chromedriver 0x0000000106ab11ce chromedriver + 860622 9 chromedriver 0x0000000106acbe76 chromedriver + 970358 10 chromedriver 0x0000000106ab0de3 chromedriver + 859619 11 chromedriver 0x0000000106a7ed7f chromedriver + 654719 12 chromedriver 0x0000000106a800de chromedriver + 659678 13 chromedriver 0x0000000106e502ad chromedriver + 4657837 14 chromedriver 0x0000000106e55130 chromedriver + 4677936 15 chromedriver 0x0000000106e5bdef chromedriver + 4705775 16 chromedriver 0x0000000106e5605a chromedriver + 4681818 17 chromedriver 0x0000000106e2892c chromedriver + 4495660 18 chromedriver 0x0000000106e73838 chromedriver + 4802616 19 chromedriver 0x0000000106e739b7 chromedriver + 4802999 20 chromedriver 0x0000000106e8499f chromedriver + 4872607 21 libsystem_pthread.dylib 0x00007ff81308d1d3 _pthread_start + 125 22 libsystem_pthread.dylib 0x00007ff813088bd3 thread_start + 15
问题分析与解决方案
核心问题
- WebDriverWait使用错误:不能将
WebElement对象传入WebDriverWait,必须用driver实例;且./div/div/svg可能不是实际可点击的元素(通常svg是图标,点击区域是其父div)。 - 循环逻辑不稳定:通过索引
div[{x}]定位产品元素,页面结构变化就会失效,应直接定位所有产品元素列表。 - 元素可见性问题:折叠元素可能在视口外,未滚动到对应位置就点击会导致无法触发。
- xpath重复错误:多个字段复用同一xpath,导致数据混乱。
- Driver资源浪费:每循环一次就创建新的Driver实例,影响性能。
修正后的代码
from selenium import webdriver from selenium.common.exceptions import NoSuchElementException, TimeoutException from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.common.by import By from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.common.action_chains import ActionChains import pandas as pd data = [] # 初始化Driver,放在循环外复用 path = '/Users/kdavid3mbp/Python/chrome_driver64/chromedriver' driver = webdriver.Chrome(path) wait = WebDriverWait(driver, 10) for y in range(1,3): website = f'https://www.knowde.com/b/markets-personal-care/products/{y}' driver.get(website) # 直接定位所有产品元素,无需索引遍历 products = wait.until(EC.presence_of_all_elements_located((By.XPATH, '//*[@id="__next"]/main/div/div[3]/div[3]/div[1]/div[2]/div'))) for product in products: try: # 滚动到产品元素位置,确保可见 ActionChains(driver).move_to_element(product).perform() # 定位可点击的展开按钮(优先父div,而非svg) expand_btn = wait.until(EC.element_to_be_clickable((By.XPATH, './/div[contains(@class, "expand-icon-container")]'))) # 检查是否已展开,避免重复点击 if 'expanded' not in product.get_attribute('class'): expand_btn.click() # 等待展开后的数据加载完成 wait.until(EC.presence_of_element_located((By.XPATH, './/a/div[2]/div/p[3]'))) except TimeoutException: # 部分产品可能默认已展开,跳过点击 pass # 提取数据,修正重复xpath问题(需根据实际页面结构调整字段路径) brand = product.find_element(By.XPATH, './/a/div[2]/div/p[1]').text item = product.find_element(By.XPATH, './/a/div[2]/div/p[2]').text inci_name = product.find_element(By.XPATH, './/a/div[2]/div/div[1]/span[2]').text # 简化元素存在性判断 ingredient_origin = product.find_element(By.XPATH, './/a/div[2]/div/div[3]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[3]/span[2]') else 'null' function = product.find_element(By.XPATH, './/a/div[2]/div/div[2]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[2]/span[2]') else 'null' benefit_claims = product.find_element(By.XPATH, './/a/div[2]/div/div[4]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[4]/span[2]') else 'null' description = product.find_element(By.XPATH, './/a/div[2]/div/p[3]').text if product.find_elements(By.XPATH, './/a/div[2]/div/p[3]') else 'null' labeling_claims = product.find_element(By.XPATH, './/a/div[2]/div/div[5]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[5]/span[2]') else 'null' compliance = product.find_element(By.XPATH, './/a/div[2]/div/div[6]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[6]/span[2]') else 'null' # 修正各字段的唯一xpath(需对照页面真实结构调整) hlb_value = product.find_element(By.XPATH, './/a/div[2]/div/div[7]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[7]/span[2]') else 'null' end_uses = product.find_element(By.XPATH, './/a/div[2]/div/div[8]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[8]/span[2]') else 'null' cas_no = product.find_element(By.XPATH, './/a/div[2]/div/div[9]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[9]/span[2]') else 'null' chemical_name = product.find_element(By.XPATH, './/a/div[2]/div/div[10]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[10]/span[2]') else 'null' synonyms = product.find_element(By.XPATH, './/a/div[2]/div/div[11]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[11]/span[2]') else 'null' chemical_family = product.find_element(By.XPATH, './/a/div[2]/div/div[12]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[12]/span[2]') else 'null' features = product.find_element(By.XPATH, './/a/div[2]/div/div[13]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[13]/span[2]') else 'null' grade = product.find_element(By.XPATH, './/a/div[2]/div/div[14]/span[2]').text if product.find_elements(By.XPATH, './/a/div[2]/div/div[14]/span[2]') else 'null' product_dict = { 'brand': brand, 'item': item, 'inci_name': inci_name, 'ingredient_origin': ingredient_origin, 'function': function, 'benefit_claims': benefit_claims, 'description': description, 'labeling_claims': labeling_claims, 'compliance': compliance, 'hlb_value': h
相关产品推荐
相关产品推荐

