使用Selenium导出链接时遭遇NoSuchDriverException及元素定位错误
批量获取PageSpeed报告链接失败:NoSuchDriverException与NoSuchElementError排查修复
问题背景
零基础借助ChatGPT编写Python脚本,拟批量从pagespeed.web.dev导出5个网站的性能报告链接。单URL测试正常,但批量执行时失败,出现NoSuchDriverException和NoSuchElementError。
原代码
from selenium import webdriver from selenium.webdriver.firefox.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time import pyperclip # For clipboard operations # Replace with the path to your GeckoDriver geckodriver_path = r'C:\Users\*****.OSHS\Documents\geckodriver.exe' # Replace with the URL of the website performance tool website_url = 'https://pagespeed.web.dev/' # Replace with the placeholder text for the input box and the export button input_box_placeholder = 'Enter a web page URL' analyze_button_xpath = '/html/body/c-wiz/div[2]/div/div[2]/form/div[2]/button/span' # Ensure this XPath correctly identifies the button copy_button_xpath = '/html/body/header/span/div[1]/button/span' # Adjust this XPath if needed # Replace with your website URL website_to_monitor = [ 'https://www.ohiostatewaterproofing.com', 'https://www.basementwaterproofing.com', 'https://www.everdrywaterproofinglouisville.com/', 'https://www.stablwall.com', 'https://www.everdrycolumbus.com' ] # Initialize the WebDriver for Firefox service = Service(executable_path=geckodriver_path) driver = webdriver.Firefox(service=service) # File to save the results results_file = 'website_performance_reports.txt' def get_report_link(driver, website_url): try: # Open the website performance tool driver.get(website_url) # Initialize WebDriverWait wait = WebDriverWait(driver, 30) # Wait for the input box by placeholder text and then find it input_box = wait.until(EC.presence_of_element_located((By.XPATH, f'//input[@placeholder="{input_box_placeholder}"]'))) input_box.send_keys(website_to_monitor) # Wait for the analyze button to be clickable and then find it analyze_button = wait.until(EC.element_to_be_clickable((By.XPATH, analyze_button_xpath))) analyze_button.click() # Wait for the analysis to complete (adjust the sleep duration as needed) time.sleep(30) # Adjust this as needed based on the website's performance # Simulate the click to copy the link copy_button_xpath = '/html/body/header/span/div[1]/button/span' # Replace with the actual XPath for the copy button copy_button = wait.until(EC.element_to_be_clickable((By.XPATH, copy_button_xpath))) copy_button.click() # Get the copied link from the clipboard report_link = pyperclip.paste() return report_link except Exception as e: print(f"Error retrieving report for {website_url}: {e}") return None def save_results(results): with open(results_file, 'w') as file: for url, report_link in results: file.write(f"Website: {url}\n") file.write(f"Report Link: {report_link}\n") file.write("----\n") def main(): # Initialize the WebDriver for Firefox service = Service(executable_path=geckodriver_path) driver = webdriver.Firefox(service=service) results = [] for website in website_to_monitor: print(f"Processing {website}...") report_link = get_report_link(driver, website) results.append((website, report_link)) # Save all results to a file save_results(results) driver.quit() print(f"All reports saved to {results_file}") if __name__ == "__main__": main()
报错信息
移除try/except后的核心报错:
File "C:\Users\*****.OSHS\OSW-Test.py", line 31, in <module> driver = webdriver.Firefox(service=service) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\*****.OSHS\AppData\Roaming\Python\Python312\site-packages\selenium\webdriver\firefox\webdriver.py", line 57, in __init__ if finder.get_browser_path(): ^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\*****.OSHS\AppData\Roaming\Python\Python312\site-packages\selenium\webdriver\common\driver_finder.py", line 47, in get_browser_path return self._binary_paths()["browser_path"] ^^^^^^^^^^^^^^^^^^^^ File "C:\Users\*****.OSHS\AppData\Roaming\Python\Python312\site-packages\selenium\webdriver\common\driver_finder.py", line 78, in _binary_paths raise NoSuchDriverException(msg) from err selenium.common.exceptions.NoSuchDriverException: Message: Unable to obtain driver for firefox; For documentation on this error, please visit: https://www.selenium.dev/documentation/webdriver/troubleshooting/errors/driver_location
执行时的元素找不到报错:
Error retrieving report for https://www.ohiostatewaterproofing.com: Message: Stacktrace: RemoteError@chrome://remote/content/shared/RemoteError.sys.mjs:8:8 WebDriverError@chrome://remote/content/shared/webdriver/Errors.sys.mjs:193:5 NoSuchElementError@chrome://remote/content/shared/webdriver/Errors.sys.mjs:511:5 dom.find/</<@chrome://remote/content/shared/DOM.sys.mjs:136:16
核心问题排查
- NoSuchDriverException:代码重复初始化Firefox Driver,全局初始化一次后main函数又初始化一次,导致驱动冲突;Selenium 4+版本需确保能正确识别Firefox安装路径。
- NoSuchElementError:
get_report_link函数错误地将整个网站列表传入输入框,而非当前循环的单个URL- 使用绝对XPath定位元素,页面结构变化时极易失效
- 固定
time.sleep(30)无法适配不同网站的加载速度,可能导致提前操作 - 重复定义
copy_button_xpath造成冗余,且定位逻辑不稳定
修复后的完整代码
from selenium import webdriver from selenium.webdriver.firefox.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time import pyperclip # GeckoDriver路径 geckodriver_path = r'C:\Users\*****.OSHS\Documents\geckodriver.exe' pagespeed_url = 'https://pagespeed.web.dev/' results_file = 'website_performance_reports.txt' # 待测试网站列表 website_to_monitor = [ 'https://www.ohiostatewaterproofing.com', 'https://www.basementwaterproofing.com', 'https://www.everdrywaterproofinglouisville.com/', 'https://www.stablwall.com', 'https://www.everdrycolumbus.com' ] def get_report_link(driver, target_url): try: driver.get(pagespeed_url) wait = WebDriverWait(driver, 40) # 等待输入框并输入目标URL input_box = wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, 'input[placeholder="Enter a web page URL"]'))) # 清空输入框避免残留内容 input_box.clear() input_box.send_keys(target_url) # 点击分析按钮(改用相对CSS定位) analyze_button = wait.until(EC.element_to_be_clickable((By.CSS_SELECTOR, 'form button[type="submit"]'))) analyze_button.click() # 等待报告生成:等待页面顶部分享按钮出现(替代固定sleep) wait.until(EC.element_to_be_clickable((By.CSS_SELECTOR, 'header button[aria-label="Share"]'))) # 额外等待几秒确保报告完全加载 time.sleep(5) # 点击复制链接按钮 copy_button = wait.until(EC.element_to_be_clickable((By.CSS_SELECTOR, 'header button[aria-label="Share"] + div button'))) copy_button.click() # 获取剪贴板中的链接 report_link = pyperclip.paste() return report_link except Exception as e: print(f"获取{target_url}报告失败: {str(e)}") return None def save_results(results): with open(results_file, 'w', encoding='utf-8') as file: for url, report_link in results: file.write(f"网站: {url}\n") file.write(f"报告链接: {report_link}\n") file.write("----\n") def main(): # 初始化Firefox Driver,添加options参数可手动指定浏览器路径 options = webdriver.FirefoxOptions() # 若Firefox安装路径非默认,取消注释并修改路径: # options.binary_location = r'C:\Program Files\Mozilla Firefox\firefox.exe' service = Service(executable_path=geckodriver_path) driver = webdriver.Firefox(service=service, options=options) results = [] for website in website_to_monitor: print(f"正在处理: {website}") report_link = get_report_link(driver, website) results.append((website, report_link)) save_results(results) driver.quit() print(f"所有报告已保存至 {results_file}") if __name__ == "__main__": main()
关键修复点
- 驱动初始化优化:仅在main函数中初始化一次Driver,避免重复创建;添加
options参数可手动指定Firefox安装路径,解决驱动识别问题 - 元素定位优化:将绝对XPath改为CSS选择器,定位逻辑更稳定,不易受页面结构变化影响
- 输入逻辑修复:传入当前循环的单个URL,添加
input_box.clear()避免输入框残留内容 - 等待逻辑优化:用等待分享按钮替代固定sleep,确保报告完全生成后再操作,减少元素找不到的概率
- 冗余代码清理:移除重复定义的变量,简化代码结构
内容的提问来源于stack exchange,提问作者TheMoonIsFake
相关产品推荐
相关产品推荐

