Selenium爬虫迭代更新DataFrame避免数据丢失的实现咨询
适配需求的爬虫实现方案
以下是直接可用的修改后代码,完全匹配你提出的所有功能要求:
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time import pandas as pd from random import randrange def crawl(input_df, save_path="./crawl_result.csv"): chrome_options = webdriver.ChromeOptions() # 不需要显示浏览器窗口的话可以打开下面的无头模式配置 # chrome_options.add_argument("--headless=new") # 初始化结果存储容器 result_list = [] query_list = input_df['Source'].unique().tolist() for idx, x in enumerate(query_list): print(f"正在爬取第{idx+1}/{len(query_list)}个源:{x}") driver = None # 预先初始化单条结果结构,保证字段永远完整 current_res = { "Source": x, "List 1": "爬取失败", "List 2": "爬取失败" } try: # 每个源单独初始化浏览器 driver = webdriver.Chrome('替换为你的chromedriver实际路径', chrome_options=chrome_options) driver.maximize_window() wait = WebDriverWait(driver, 30) driver.get('替换为你的爬取链接前缀/'+x) time.sleep(randrange(5)) driver.execute_script("window.scrollTo(0, 1000)") # 爬取List1字段 my1 = wait.until(EC.visibility_of_element_located((By.XPATH, "//div[text()='Trustscore']/../following-sibling::div/descendant::div[@class='icon']"))).text current_res["List 1"] = my1 # 爬取List2字段 try: my2 = wait.until(EC.visibility_of_element_located((By.XPATH, "//div[text()='Company data']/../following-sibling::div/descendant::b[text()='Alexa rank']/../following-sibling::div"))).text current_res["List 2"] = my2 except: current_res["List 2"] = "Data not available" except Exception as e: print(f"源{x}爬取出错,错误信息:{str(e)}") finally: # 无论爬取成功失败都强制关闭当前浏览器 if driver: driver.quit() # 追加当前源的结果到总列表 result_list.append(current_res) # 即时写入本地CSV,彻底避免中途崩溃丢失数据 temp_df = pd.DataFrame(result_list) temp_df.to_csv(save_path, index=False, encoding="utf-8-sig") # 非最后一个源的话,间隔15秒再爬下一个 if idx != len(query_list) - 1: time.sleep(15) # 最终返回完整的结果DataFrame final_df = pd.DataFrame(result_list) return final_df
核心改动说明
- 单源独立浏览器生命周期:将浏览器初始化、关闭逻辑放到每个Source的遍历循环内部,每个源爬取前启动Chrome,爬取完成后无论成功失败都强制关闭浏览器,符合需求
- 彻底解决数组长度不一致错误:每个Source对应固定结构的单条结果,无论爬取成功失败都会将对应结果追加到总列表,三个字段永远一一对应,不会出现长度不匹配的报错
- 即时数据持久化:每完成一个Source的爬取,就将当前已爬取的所有结果写入本地CSV文件,就算程序中途崩溃,已完成的爬取数据都完整保存在本地,不会丢失
- 单源错误不中断整体流程:错误捕获范围仅针对当前Source的爬取逻辑,单个源爬取失败时仅对应该源的字段填入错误提示,不会打断剩余源的爬取任务
- 爬取间隔符合要求:两个Source的爬取请求之间固定休眠15秒,降低被反爬的概率
内容的提问来源于stack exchange,提问作者LdM
相关产品推荐
相关产品推荐

