Selenium爬取含弹窗院校页面:无法点击院校链接求助
解决WHED网站院校信息爬取的链接点击问题
我正在爬取网站https://www.whed.net/results_institutions.php,目前已能从下拉框选择国家并点击OK获取结果,但页面中的院校链接无法点击,无法获取院校名称、网址(WWW)及城市信息。以下是针对阿富汗的实现代码,该代码可初始化Selenium驱动,但无法完成院校链接点击操作,请求技术帮助:
service = Service("C:/Selenium_drivers/chromedriver-win64/chromedriver.exe") driver = webdriver.Chrome(service=service) driver.get(url) country = 'Afghanistan' institues = [] cities = [] wwws = [] drop_down = Select(driver.find_element(By.XPATH, '//select')) drop_down.select_by_visible_text(country) all_institute = driver.find_element(By.XPATH, "//input[@id='membre2']") if not all_institute.is_selected(): all_institute.click() button = driver.find_element(By.XPATH, "//input[@type='button']") button.click() results_per_page = Select(driver.find_element(By.XPATH, "//select[@name='nbr_ref_pge']")) results_per_page.select_by_visible_text('100') total_results = int(driver.find_element(By.XPATH, "//p[@class='infos']").text.split()[0]) max_iter = total_results//100 + 1 iterations = 0 go_on = True while go_on: iterations += 1 institutions = driver.find_elements(By.XPATH, "//li[contains(@class, 'clearfix plus')]") for institue in institutions: link = institute.find_element(By.XPATH, ".//h3/a") link.click() time.sleep(2) pop_up = driver.find_element(By.XPATH, "//iframe[starts-with(@id, 'fancybox-frame')]") driver.switch_to_frame(pop_up) # main_window = driver.current_window_handle # Store the handle of the main window # popup_window = None # for window_handle in driver.window_handles: # if window_handle != main_window: # popup_window = window_handle # Switch to the popup window # driver.switch_to.window(popup_window) institue = driver.find_element(By.XPATH, "//div[@class='detail_right']/div[1]").text city = driver.find_element(By.XPATH, "//span[@class='libelle' and text() = 'City:']/following-sibling::span[@class='contenu']").text www = driver.find_element(By.XPATH, "//span[@class='libelle' and text() = 'WWW:']/following-sibling::span[@class='contenu']").get_attribute("title") institues.append(institute) cities.append(city) wwws.append(www) close_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//a[@title='Close']"))) close_button.click() # driver.switch_to.window(main_window) # driver.switch_to.window(main_window) if iterations >= max_iter: go_on =False break time.sleep(2) next_page = driver.find_elements(By.XPATH, "//a[@title='Next page' ]")[0] next_page.click()
问题排查与修正方案
你的代码存在几个关键问题导致链接无法点击、信息获取失败,以下是修正后的完整代码及关键修改说明:
修正后的代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import Select, WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.chrome.service import Service import time service = Service("C:/Selenium_drivers/chromedriver-win64/chromedriver.exe") driver = webdriver.Chrome(service=service) url = "https://www.whed.net/results_institutions.php" driver.get(url) country = 'Afghanistan' institutes = [] cities = [] wwws = [] # 选择国家并提交 drop_down = Select(driver.find_element(By.XPATH, '//select')) drop_down.select_by_visible_text(country) all_institute = driver.find_element(By.XPATH, "//input[@id='membre2']") if not all_institute.is_selected(): all_institute.click() button = driver.find_element(By.XPATH, "//input[@type='button']") button.click() # 设置每页显示100条结果 results_per_page = Select(driver.find_element(By.XPATH, "//select[@name='nbr_ref_pge']")) results_per_page.select_by_visible_text('100') # 获取总结果数并计算分页次数 total_results = int(driver.find_element(By.XPATH, "//p[@class='infos']").text.split()[0]) max_iter = total_results // 100 + 1 iterations = 0 go_on = True wait = WebDriverWait(driver, 10) # 初始化显式等待 while go_on: iterations += 1 # 等待当前页院校列表加载完成 institutions = wait.until(EC.presence_of_all_elements_located((By.XPATH, "//li[contains(@class, 'clearfix plus')]"))) for institute in institutions: try: # 等待院校链接可点击并点击 link = wait.until(EC.element_to_be_clickable((By.XPATH, ".//h3/a"), root=institute)) link.click() # 等待弹窗iframe加载完成并切换 pop_up = wait.until(EC.presence_of_element_located((By.XPATH, "//iframe[starts-with(@id, 'fancybox-frame')]"))) driver.switch_to.frame(pop_up) # 获取院校信息 institute_name = wait.until(EC.presence_of_element_located((By.XPATH, "//div[@class='detail_right']/div[1]"))).text city = wait.until(EC.presence_of_element_located((By.XPATH, "//span[@class='libelle' and text()='City:']/following-sibling::span[@class='contenu']"))).text www_element = wait.until(EC.presence_of_element_located((By.XPATH, "//span[@class='libelle' and text()='WWW:']/following-sibling::span[@class='contenu']"))) www = www_element.get_attribute("title") institutes.append(institute_name) cities.append(city) wwws.append(www) # 关闭弹窗并切回主页面 close_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//a[@title='Close']"))) close_button.click() driver.switch_to.default_content() # 切回主上下文 time.sleep(1) # 短暂等待页面恢复 except Exception as e: print(f"处理院校时出错: {e}") driver.switch_to.default_content() # 出错后确保切回主页面 continue if iterations >= max_iter: go_on = False break # 点击下一页 next_page = wait.until(EC.element_to_be_clickable((By.XPATH, "//a[@title='Next page']"))) next_page.click() time.sleep(2) # 等待下一页加载 # 输出结果 print("院校名称:", institutes) print("城市:", cities) print("网址:", wwws) driver.quit()
关键修改说明
- 修复变量拼写错误:原代码中循环变量
institue与内部使用的institute不一致,导致找不到元素,统一修正为institute - 添加显式等待:替换不稳定的
time.sleep,使用WebDriverWait等待元素加载/可点击,避免页面未加载完成就执行操作 - 更新iframe切换方法:将过时的
driver.switch_to_frame改为driver.switch_to.frame,并在操作完成后用driver.switch_to.default_content()切回主页面上下文 - 增强元素定位可靠性:所有元素获取都添加等待,确保元素存在后再操作
- 异常处理:添加try-except块捕获异常,避免单个院校处理失败导致整个程序崩溃,出错后强制切回主页面上下文
- 初始化显式等待对象:提前创建
WebDriverWait实例,复用等待配置
内容的提问来源于stack exchange,提问作者Aditya Maurya
相关产品推荐
相关产品推荐

