如何用Python+Selenium读取CSV中ID并批量爬取网站数据至CSV
批量爬取用户数据并保存至CSV的实现方案
核心改进方向
- 用
csv模块批量读取ID列表,替代手动输入单个ID - 改用显式等待处理页面加载,确保数据完全加载后再爬取
- 实现循环处理逻辑,控制爬取1000个用户,同时将结果写入新CSV
完整实现代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import csv # 初始化浏览器 DRIVER_PATH = '/chromedriver' driver = webdriver.Chrome(executable_path=DRIVER_PATH) driver.get('https://g12-result.moe.gov.eg/') driver.maximize_window() # 读取ID列表CSV ids = [] with open('/Id.csv', 'r', newline='', encoding='utf-8') as csv_file: reader = csv.reader(csv_file) # 若你的CSV有表头,取消注释下面这行跳过表头 # next(reader) for row in reader: ids.append(row[0]) # 假设ID在CSV的第一列 # 准备保存结果的CSV with open('/user_results.csv', 'w', newline='', encoding='utf-8') as result_file: writer = csv.writer(result_file) # 写入结果表头(请根据页面实际字段调整) writer.writerow(['用户ID', '姓名', '总分', '各科成绩']) # 控制爬取1000个用户 count = 0 for user_id in ids: if count >= 1000: break try: # 等待输入框出现,清空后输入ID id_input = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.ID, "SeatingNo")) ) id_input.clear() id_input.send_keys(user_id) # 等待提交按钮可点击,然后点击 submit_btn = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.XPATH, "//button[@type='submit']")) ) submit_btn.click() # 等待结果页面加载完成(替换成页面上真实的结果元素选择器,比如某个成绩容器) WebDriverWait(driver, 15).until( EC.presence_of_element_located((By.CLASS_NAME, "result-content")) # 示例选择器,需自行修改 ) # 提取数据(根据页面实际HTML结构调整元素选择器) name = driver.find_element(By.XPATH, "//span[@id='student-name']").text total_score = driver.find_element(By.XPATH, "//div[@class='total-mark']").text subject_scores = [elem.text for elem in driver.find_elements(By.XPATH, "//td[@class='subject-mark']")] subjects_str = ', '.join(subject_scores) # 写入结果到CSV writer.writerow([user_id, name, total_score, subjects_str]) count += 1 print(f"已完成第{count}个用户爬取:{user_id}") # 返回查询页面,准备下一个ID查询 driver.back() except Exception as e: print(f"处理ID {user_id}时出错:{str(e)}") # 出错后返回查询页面,避免程序卡住 driver.back() continue # 关闭浏览器 driver.quit()
关键注意事项
- 显式等待:相比全局的
implicitly_wait,显式等待针对每个操作等待特定元素,能有效避免网络延迟导致的元素找不到问题 - CSV处理:用
utf-8编码避免乱码,若ID CSV包含表头,记得启用next(reader)跳过 - 元素选择器:代码中的结果提取选择器(如
student-name、total-mark)是示例,必须根据目标页面的真实HTML结构修改 - 异常处理:捕获单个ID的处理错误,避免整个程序崩溃,同时出错后自动返回查询页面继续处理
- 爬取数量控制:通过计数器
count精准控制只爬取1000个用户,达到数量后自动终止循环
内容的提问来源于stack exchange,提问作者Asma Ahmed
相关产品推荐
相关产品推荐

