WHED高校网站爬虫空DataFrame问题排查与数据采集实现
WHED全球院校爬虫空结果问题修复方案
问题根因
- Selenium版本兼容问题:原有代码使用的
find_elements_by_xpath/find_element_by_id等旧版API,在Selenium 4.0及以上版本已被完全移除,调用直接抛出异常中断流程,最终返回空DataFrame - 元素定位失效:WHED官网前端结构迭代后,原有代码写死的部分XPath路径(如院校标题、国家字段路径)已无法匹配当前页面DOM结构
- 国家列表逻辑错误:初始写死的国家选项包含不存在的分组占位值,同时采集下拉选项时把optgroup分组标签也加入遍历列表,选择不存在的选项时直接触发选择异常
- 等待逻辑缺失:点击查询、翻页后未等待页面加载完成就抓取元素,经常捕获到旧页面空内容
- 翻页逻辑缺陷:点击下一页后未做页面加载校验,容易出现重复采集、漏采问题
- 无容错和断点续传:单个页面加载失败就会中断全量爬取,中途异常退出后已爬数据直接丢失
前置准备
- 安装依赖:执行
pip install pandas selenium安装所需库 - 注意:本地安装正式版Chrome浏览器即可,Selenium 4+内置驱动管理器,会自动匹配对应版本ChromeDriver,不需要手动下载指定路径
修复后可直接运行代码
import time import pandas as pd from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import Select from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.common.exceptions import NoSuchElementException, TimeoutException begin = time.time() result = [] univ_links = [] target_fields = ['Street:', 'City:', 'Province:', 'Post Code:', 'WWW:', 'Fields of study:', 'Job title:'] # 初始化浏览器配置,规避反爬检测 options = webdriver.ChromeOptions() options.add_argument('--disable-blink-features=AutomationControlled') options.add_experimental_option("excludeSwitches", ["enable-automation"]) webD = webdriver.Chrome(options=options) wait = WebDriverWait(webD, 10) # 进入院校搜索页 webD.get("https://www.whed.net/results_institutions.php") wait.until(EC.presence_of_element_located((By.ID, 'Chp1'))) # 采集所有有效国家/地区选项,过滤分组占位项 country_select = Select(webD.find_element(By.ID, 'Chp1')) countries = [] for opt in country_select.options: if opt.get_attribute('value') != '' and opt.text.strip() != '': countries.append(opt.text.strip()) print(f"共加载{len(countries)}个国家/地区的待爬列表") # 遍历所有国家采集院校详情链接 for idx, cntry in enumerate(countries): print(f"正在处理第{idx+1}/{len(countries)}个地区:{cntry}") # 选择当前国家 country_select = Select(wait.until(EC.presence_of_element_located((By.ID, 'Chp1')))) country_select.select_by_visible_text(cntry) # 点击查询按钮 webD.find_element(By.XPATH, '//*[@id="fsearch"]/p/input').click() # 设置每页展示100条结果,减少翻页次数 try: rpp_select = Select(wait.until(EC.presence_of_element_located((By.NAME, 'nbr_ref_pge')))) rpp_select.select_by_visible_text('100') time.sleep(1) except TimeoutException: print(f"当前地区无院校数据,跳过") continue # 循环翻页采集所有院校链接 while True: wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, '#results li'))) university_list = webD.find_elements(By.CSS_SELECTOR, '#results li') for univ in university_list: try: href = univ.find_element(By.CSS_SELECTOR, '.details a').get_attribute('href') univ_links.append(href) except NoSuchElementException: continue # 检查是否存在下一页按钮 try: next_btn = webD.find_element(By.PARTIAL_LINK_TEXT, 'Next') next_btn.click() time.sleep(1.5) except NoSuchElementException: break print(f"共采集到{len(univ_links)}条院校详情链接,开始抓取详情信息") # 加载已爬数据做断点续爬,避免异常退出后重复采集 try: exist_data = pd.read_csv("WHED全球院校数据.csv") crawled_links = exist_data['详情页链接'].tolist() except: exist_data = pd.DataFrame() crawled_links = [] # 遍历采集院校详情 field_map = { 'Street:': '街道地址', 'City:': '城市', 'Province:': '省/州', 'Post Code:': '邮编' } for link_idx, detail_url in enumerate(univ_links): if detail_url in crawled_links: continue print(f"正在抓取第{link_idx+1}/{len(univ_links)}条院校数据") try: webD.get(detail_url) wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, '.ficheEtablissement'))) # 采集基础信息 univ_name = webD.find_element(By.XPATH, '//*[@id="contenu"]/h1').text.strip() country = webD.find_element(By.XPATH, '//*[@id="contenu"]/p[@class="pays"]').text.strip() temp = { '院校名称': univ_name, '所属国家/地区': country, '详情页链接': detail_url } # 采集结构化字段 libelles = webD.find_elements(By.CLASS_NAME, 'libelle') contenus = webD.find_elements(By.CLASS_NAME, 'contenu') study_fields = '' job_titles = '' for j in range(len(libelles)): label = libelles[j].text.strip() content_val = contenus[j].text.strip().replace('\n', ', ') if label == 'WWW:': temp['官方网站'] = content_val elif label == 'Fields of study:': study_fields += content_val + ',' elif label == 'Job title:': job_titles += content_val + ',' elif label in target_fields: temp[field_map[label]] = content_val temp['开设学科领域'] = study_fields.rstrip(',') temp['相关负责人头衔'] = job_titles.rstrip(',') result.append(temp) # 每爬10条自动存盘,防止数据丢失 if len(result) % 10 == 0: temp_df = pd.DataFrame(result) full_df = pd.concat([exist_data, temp_df], ignore_index=True) full_df.to_csv("WHED全球院校数据.csv", index=False, encoding='utf-8-sig') except Exception as e: print(f"链接{detail_url}抓取失败,跳过,异常信息:{str(e)}") continue # 最终数据去重、存盘 final_df = pd.DataFrame(result) full_data = pd.concat([exist_data, final_df], ignore_index=True) full_data = full_data.drop_duplicates(subset=['详情页链接'], keep='last') full_data.to_csv("WHED全球院校数据.csv", index=False, encoding='utf-8-sig') webD.quit() end = time.time() print(f"爬取完成,总耗时{round(end-begin, 2)}秒,共采集有效院校数据{len(full_data)}条")
数据使用说明
- 导出的CSV文件采用
utf-8-sig编码,直接用Excel打开不会出现中文乱码 - 作为Web端院校下拉选项数据源时,直接取「院校名称」「所属国家/地区」字段做联动即可,「官方网站」「开设学科领域」字段可作为补充信息
- 投入使用前可按需求过滤名称为空、官网无效的异常条目,按国家、院校名称排序后即可直接接入业务系统
内容的提问来源于stack exchange,提问作者Ab900
相关产品推荐
相关产品推荐

