Selenium爬取网页生成DataFrame的404/无数据场景条件判断咨询
实现方案
完整可运行代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import pandas as pd # 待处理站点列表 site_list = ['stackoverflow.com', 'livevsfox.ca'] # 提前初始化存储列 webs = [] countries = [] organisations = [] # 公共配置,替换为自己的实际参数 QUERY_BASE_URL = "替换为你实际的查询站点前缀" CHROME_DRIVER_PATH = "替换为你的chromedriver本地路径" for site in site_list: # 每个站点单独初始化driver,避免缓存、会话污染,适配单站点处理要求 chrome_options = webdriver.ChromeOptions() driver = webdriver.Chrome(executable_path=CHROME_DRIVER_PATH, options=chrome_options) wait = WebDriverWait(driver, 15) # 默认值统一设为Missing country = 'Missing' org = 'Missing' try: driver.get(f"{QUERY_BASE_URL}{site}") driver.execute_script("window.scrollTo(0, 1000)") # 优先检测404错误提示 try: error_h2 = wait.until(EC.visibility_of_element_located((By.CSS_SELECTOR, "section.selection div.container h2")), timeout=5) if "could not be found or reached" in error_h2.text: # 命中404直接使用默认值,跳过提取逻辑 pass except: # 无404提示,进入正常提取流程 # 提取国家信息 try: country_ele = wait.until(EC.visibility_of_element_located( (By.XPATH, "//div[text()='Company data']/../following-sibling::div/descendant::b[text()='Country']/../following-sibling::div") ), timeout=10) country = country_ele.text.strip() except: country = 'Missing' # 提取机构信息 try: org_ele = wait.until(EC.visibility_of_element_located( (By.XPATH, "//div[text()='Company data']/../following-sibling::div/descendant::b[text()='Organisation']/../following-sibling::div") ), timeout=10) org = org_ele.text.strip() except: org = 'Missing' finally: # 无论是否出现异常,强制关闭浏览器,避免残留进程 driver.quit() # 存入结果列表 webs.append(site) countries.append(country) organisations.append(org) # 生成最终DataFrame df = pd.DataFrame({ 'WEB': webs, 'Country': countries, 'Organisation': organisations }) print(df)
逻辑说明
- 每个站点单独启动、关闭浏览器,完全规避批量操作触发验证码的问题,也不会出现页面缓存影响提取结果的情况
- 优先检测404提示元素,命中后直接使用默认
Missing值,不需要执行后续提取逻辑 - 所有元素提取操作都加了异常捕获,任意字段提取失败都会统一填充
Missing,不会出现DataFrame列长度不一致的报错 - 用
finally块强制关闭浏览器,避免异常场景下浏览器后台残留
内容的提问来源于stack exchange,提问作者LdM
相关产品推荐
相关产品推荐

