Python Selenium爬虫翻页报错NoSuchElementException无法定位h2元素求助
Selenium爬取kotsovolos.gr报错解决方案
报错原因
- 翻页后页面未完全加载就执行元素查找,目标h2元素还没渲染完成
- 部分产品卡片结构特殊,内部不存在h2标签,未做容错处理直接查找就会抛出异常
- 固定等待时间不可靠,网络波动时页面加载速度慢就会触发元素查找失败
修改方案
1. 引入显式等待和异常捕获机制
首先在代码头部导入对应依赖:
from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.common.exceptions import NoSuchElementException, TimeoutException
2. 优化核心爬取循环逻辑
把原代码中的while循环部分替换为如下内容:
while True: # 等待产品卡片加载完成 try: WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.CSS_SELECTOR, 'div.product')) ) except TimeoutException: # 10秒未加载出产品直接退出循环 break storage_box = driver.find_elements_by_css_selector('div.product') for storage_boxes in storage_box: try: # 查找元素失败就跳过当前卡片 product_url = storage_boxes.find_element_by_tag_name('h2') product_urls = product_url.find_element_by_tag_name('a').get_attribute('href') print(product_urls) p_links.append(product_urls) p_model = storage_boxes.find_element_by_css_selector('div.title a').text print(p_model) models.append(p_model) manufacturer1 = p_model.split(" ") print(manufacturer1[0]) titles.append(manufacturer1[0]) memory = re.findall('\d+ ?[gG][bB]',p_model) print(memory) memory1 = str(memory).replace("['",'').replace("']",'').replace("[]",'').strip() if "," in memory1: arr=memory1.split(",") memory_str = 'N/A' for str1 in arr: str2=str1.replace("GB", "").replace("gb", "").replace("'", "").strip() if len(str2)!=1: memory_str=str1 break elif (memory1 == ""): memory_str ='N/A' else: memory_str=memory1 memory_str = memory_str.replace("'", "").strip() print(memory_str) memorys.append(memory_str) colors= [] prod_color = p_model.split(" ") length = len(prod_color) indexcolor = length-3 colors.append(prod_color[indexcolor]) color1 = str(colors).replace("['",'').replace("']",'').strip() print(color1) p_colors.append(color1) p_price = storage_boxes.find_element_by_css_selector('.priceWithVat > .price').text print(p_price) prod_prices.append(p_price) except NoSuchElementException: # 跳过缺少必要元素的异常卡片 continue # 判断是否存在下一页 try: next_btn = driver.find_element_by_css_selector('.pagination_next a') next_url = next_btn.get_attribute('href') driver.get(next_url) time.sleep(2) except NoSuchElementException: # 没有下一页直接退出循环 break
3. 补充Excel数据写入逻辑
原代码仅初始化了Excel表头,未写入爬取到的实际数据,在循环结束后添加如下代码:
# 批量写入爬取到的所有数据 for index in range(len(p_links)): row = index + 1 ws.write(row, 0, titles[index]) ws.write(row, 1, p_links[index]) ws.write(row, 2, prod_prices[index]) ws.write(row, 3, models[index]) ws.write(row, 4, memorys[index]) ws.write(row, 5, self.currency) ws.write(row, 6, p_colors[index]) ws.write(row, 7, self.VAT) ws.write(row, 8, self.shipping) ws.write(row, 9, self.Pre_PromotionPrice) ws.write(row, 10, self.country) ws.write(row, 11, str(today)) ws.write(row, 12, models[index]) wb.save(r"C:\Users\Karthick R\Desktop\VS code\kotsovolos.xls") # 爬取完成关闭浏览器 driver.quit()
内容的提问来源于stack exchange,提问作者vinitha
相关产品推荐
相关产品推荐

