Python Selenium:遍历表格列、访问URL并爬取内页内容故障排查
Selenium爬取表格后续行时出现NoSuchElementException的解决方案
问题回顾
遍历表格列找到指定数字的精确匹配行后,需要爬取该行后续所有行的内页数据,但返回表格页后无法定位元素,抛出NoSuchElementException。尝试了driver.back()和新标签页两种方式均未解决。
问题根源
- 逻辑冗余导致页面状态混乱:代码同时执行了
bak_element.click()(原页面跳转)和新标签页打开操作,原页面已跳转到内页,后续driver.back()返回表格页后,页面可能重新渲染,旧元素引用全部失效。 - 依赖静态元素列表:初始获取的
table_rows是静态列表,页面状态变化后元素引用过期,循环仍基于初始列表长度,易出现索引不匹配。 - 过度依赖固定等待:
time.sleep不可靠,可能导致元素未加载完成就尝试定位,或浪费不必要的时间。
修复后的完整代码
from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import chromedriver_autoinstaller from selenium.common.exceptions import ( TimeoutException, NoSuchElementException, StaleElementReferenceException ) import undetected_chromedriver as uc class InitiateRGM: def __init__(self): self.driver = self.setup_driver() self.doc_selection = None self.bak_no_text = "123456" # 目标匹配数字 def setup_driver(self): options = uc.ChromeOptions() options.add_argument("--start-maximized") chromedriver_autoinstaller.install() return uc.Chrome(options=options) def open_rgm(self): self.driver.get("https://website-to-scrape.com/") input("Press Enter to start... ") WebDriverWait(self.driver, 10).until(EC.presence_of_element_located((By.ID, "btn-requests"))) def scrape_table(self): # 打开请求标签页并等待表格加载 open_requests_tab = WebDriverWait(self.driver, 30).until( EC.element_to_be_clickable((By.XPATH, '//*[@id="btn-requests"]/span[1]')) ) self.driver.execute_script("arguments[0].scrollIntoView(true);", open_requests_tab) open_requests_tab.click() # 等待表格完全加载 WebDriverWait(self.driver, 30).until( EC.presence_of_element_located((By.XPATH, '//*[@id="table-requests"]/table/tbody/tr')) ) base_bak_xpath = '//*[@id="table-requests"]/table/tbody/tr' found_target = False # 动态获取行数,避免依赖静态列表 while True: try: # 每次循环重新获取所有行,避免stale元素 all_rows = WebDriverWait(self.driver, 10).until( EC.presence_of_all_elements_located((By.XPATH, f"{base_bak_xpath}/td[2]/a")) ) if not all_rows: break # 没有更多行时退出 for index, row_element in enumerate(all_rows): try: bak_no = row_element.get_attribute("innerHTML").strip() # 找到目标行后标记开始爬取后续行 if bak_no == self.bak_no_text: found_target = True continue if found_target: # 获取内页URL,直接在新标签页打开 inner_url = f"https://website-to-scrape/#request/{bak_no}" self.driver.execute_script(f"window.open('{inner_url}', '_blank');") # 切换到新标签页 self.driver.switch_to.window(self.driver.window_handles[-1]) # 爬取内页数据 try: detail_element = WebDriverWait(self.driver, 20).until( EC.presence_of_element_located((By.XPATH, '//*[@id="info-main"]/table/tbody/tr[5]/td[2]')) ) detail_text = detail_element.get_attribute("innerHTML").strip() print(f"获取到数据: {detail_text}") except (TimeoutException, NoSuchElementException) as e: print(f"爬取内页{bak_no}失败: {e}") # 关闭新标签页并切回原标签页 self.driver.close() self.driver.switch_to.window(self.driver.window_handles[0]) # 移除已处理的行(避免重复处理) all_rows = all_rows[index+1:] break # 跳出当前循环,重新获取剩余行 except StaleElementReferenceException: # 元素过期则重新获取所有行,继续循环 break else: # 所有行处理完毕,退出循环 break except TimeoutException: print("表格加载超时,退出") break if __name__ == "__main__": scraper = InitiateRGM() scraper.open_rgm() scraper.scrape_table() scraper.driver.quit()
关键改进点
- 清理冗余跳转逻辑:仅保留新标签页打开内页的操作,原表格页始终保持在当前标签页,避免页面状态变化导致的元素失效。
- 动态获取表格行:每次循环重新获取所有行,解决元素引用过期(
StaleElementReferenceException)问题。 - 替换固定等待为显式等待:使用
WebDriverWait等待元素可点击/可见,确保操作可靠性。 - 捕获元素过期异常:处理
StaleElementReferenceException,重新获取元素后继续执行。 - 避免重复处理:处理完一行后截断行列表,只保留未处理部分,提升执行效率。
内容的提问来源于stack exchange,提问作者Daniel M
相关产品推荐
相关产品推荐

