You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Selenium爬取含弹窗院校页面:无法点击院校链接求助

解决WHED网站院校信息爬取的链接点击问题

我正在爬取网站https://www.whed.net/results_institutions.php,目前已能从下拉框选择国家并点击OK获取结果,但页面中的院校链接无法点击,无法获取院校名称、网址(WWW)及城市信息。以下是针对阿富汗的实现代码,该代码可初始化Selenium驱动,但无法完成院校链接点击操作,请求技术帮助:

service = Service("C:/Selenium_drivers/chromedriver-win64/chromedriver.exe")

driver = webdriver.Chrome(service=service)

driver.get(url)

country = 'Afghanistan'

institues = []
cities = []
wwws = []

drop_down = Select(driver.find_element(By.XPATH, '//select'))
drop_down.select_by_visible_text(country)

all_institute = driver.find_element(By.XPATH, "//input[@id='membre2']")
if not all_institute.is_selected():
    all_institute.click()
    
button = driver.find_element(By.XPATH, "//input[@type='button']")

button.click()


results_per_page = Select(driver.find_element(By.XPATH, "//select[@name='nbr_ref_pge']"))
results_per_page.select_by_visible_text('100')


total_results = int(driver.find_element(By.XPATH, "//p[@class='infos']").text.split()[0])

max_iter = total_results//100 + 1
iterations = 0

go_on = True

while go_on:
    iterations += 1
    
    institutions = driver.find_elements(By.XPATH, "//li[contains(@class, 'clearfix plus')]")
    
    
    for institue in institutions:

            link = institute.find_element(By.XPATH, ".//h3/a")
            link.click()
            
            time.sleep(2)
            
            pop_up = driver.find_element(By.XPATH, "//iframe[starts-with(@id, 'fancybox-frame')]")
            
            driver.switch_to_frame(pop_up)

    #             main_window = driver.current_window_handle  # Store the handle of the main window
    #             popup_window = None

    #             for window_handle in driver.window_handles:
    #                 if window_handle != main_window:
    #                     popup_window = window_handle

            # Switch to the popup window
    #             driver.switch_to.window(popup_window)

            institue = driver.find_element(By.XPATH, "//div[@class='detail_right']/div[1]").text

            city = driver.find_element(By.XPATH, "//span[@class='libelle' and text() = 'City:']/following-sibling::span[@class='contenu']").text

            www = driver.find_element(By.XPATH, "//span[@class='libelle' and text() = 'WWW:']/following-sibling::span[@class='contenu']").get_attribute("title")

            institues.append(institute)

            cities.append(city)

            wwws.append(www)

            close_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//a[@title='Close']")))
            close_button.click()

#             driver.switch_to.window(main_window)

#             driver.switch_to.window(main_window)

    if iterations >= max_iter:
        go_on =False
        break
        
    time.sleep(2)
    
    next_page = driver.find_elements(By.XPATH, "//a[@title='Next page' ]")[0]
    next_page.click()

问题排查与修正方案

你的代码存在几个关键问题导致链接无法点击、信息获取失败,以下是修正后的完整代码及关键修改说明:

修正后的代码

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import Select, WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.chrome.service import Service
import time

service = Service("C:/Selenium_drivers/chromedriver-win64/chromedriver.exe")
driver = webdriver.Chrome(service=service)
url = "https://www.whed.net/results_institutions.php"
driver.get(url)

country = 'Afghanistan'
institutes = []
cities = []
wwws = []

# 选择国家并提交
drop_down = Select(driver.find_element(By.XPATH, '//select'))
drop_down.select_by_visible_text(country)

all_institute = driver.find_element(By.XPATH, "//input[@id='membre2']")
if not all_institute.is_selected():
    all_institute.click()

button = driver.find_element(By.XPATH, "//input[@type='button']")
button.click()

# 设置每页显示100条结果
results_per_page = Select(driver.find_element(By.XPATH, "//select[@name='nbr_ref_pge']"))
results_per_page.select_by_visible_text('100')

# 获取总结果数并计算分页次数
total_results = int(driver.find_element(By.XPATH, "//p[@class='infos']").text.split()[0])
max_iter = total_results // 100 + 1
iterations = 0
go_on = True

wait = WebDriverWait(driver, 10)  # 初始化显式等待

while go_on:
    iterations += 1
    # 等待当前页院校列表加载完成
    institutions = wait.until(EC.presence_of_all_elements_located((By.XPATH, "//li[contains(@class, 'clearfix plus')]")))
    
    for institute in institutions:
        try:
            # 等待院校链接可点击并点击
            link = wait.until(EC.element_to_be_clickable((By.XPATH, ".//h3/a"), root=institute))
            link.click()
            
            # 等待弹窗iframe加载完成并切换
            pop_up = wait.until(EC.presence_of_element_located((By.XPATH, "//iframe[starts-with(@id, 'fancybox-frame')]")))
            driver.switch_to.frame(pop_up)
            
            # 获取院校信息
            institute_name = wait.until(EC.presence_of_element_located((By.XPATH, "//div[@class='detail_right']/div[1]"))).text
            city = wait.until(EC.presence_of_element_located((By.XPATH, "//span[@class='libelle' and text()='City:']/following-sibling::span[@class='contenu']"))).text
            www_element = wait.until(EC.presence_of_element_located((By.XPATH, "//span[@class='libelle' and text()='WWW:']/following-sibling::span[@class='contenu']")))
            www = www_element.get_attribute("title")
            
            institutes.append(institute_name)
            cities.append(city)
            wwws.append(www)
            
            # 关闭弹窗并切回主页面
            close_button = wait.until(EC.element_to_be_clickable((By.XPATH, "//a[@title='Close']")))
            close_button.click()
            driver.switch_to.default_content()  # 切回主上下文
            
            time.sleep(1)  # 短暂等待页面恢复
        except Exception as e:
            print(f"处理院校时出错: {e}")
            driver.switch_to.default_content()  # 出错后确保切回主页面
            continue
    
    if iterations >= max_iter:
        go_on = False
        break
    
    # 点击下一页
    next_page = wait.until(EC.element_to_be_clickable((By.XPATH, "//a[@title='Next page']")))
    next_page.click()
    time.sleep(2)  # 等待下一页加载

# 输出结果
print("院校名称:", institutes)
print("城市:", cities)
print("网址:", wwws)

driver.quit()

关键修改说明

  • 修复变量拼写错误:原代码中循环变量institue与内部使用的institute不一致,导致找不到元素,统一修正为institute
  • 添加显式等待:替换不稳定的time.sleep,使用WebDriverWait等待元素加载/可点击,避免页面未加载完成就执行操作
  • 更新iframe切换方法:将过时的driver.switch_to_frame改为driver.switch_to.frame,并在操作完成后用driver.switch_to.default_content()切回主页面上下文
  • 增强元素定位可靠性:所有元素获取都添加等待,确保元素存在后再操作
  • 异常处理:添加try-except块捕获异常,避免单个院校处理失败导致整个程序崩溃,出错后强制切回主页面上下文
  • 初始化显式等待对象:提前创建WebDriverWait实例,复用等待配置

内容的提问来源于stack exchange,提问作者Aditya Maurya

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.08 02:17:20