Selenium+BeautifulSoup抓取企业动态内容失败问题求助
代码优化建议及修正版本
核心优化点
- 复用浏览器实例:原代码每次请求都启动新Chrome实例,效率极低且易触发反爬,改为全局复用一个浏览器
- 替换固定sleep为显式等待:等待目标元素加载完成再解析,彻底解决网络延迟导致的元素未加载问题
- 优化CSS选择器精度:原选择器过于宽泛(如
div.flex.flex-row.items-center span),新增文本校验或父容器限定,避免匹配错误元素 - 修复BeautifulSoup语法问题:
html.parser不支持:contains()选择器,改用遍历匹配文本的方式提取Founded/Team Size等字段 - 避免变量名冲突:函数内的
url变量覆盖入参,重命名为company_url - 增加重试机制:针对临时网络错误或反爬拦截,自动重试2-3次
- 细化异常处理:区分不同错误类型,便于定位问题
- 增强防反爬配置:模拟真实UA、启用新版无头模式,降低被识别为爬虫的概率
修正后的完整代码
from selenium import webdriver from selenium.webdriver.chrome.options import Options from selenium.webdriver.chrome.service import Service from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from bs4 import BeautifulSoup import pandas as pd import time import random # 配置Chrome选项 options = Options() options.add_argument("--headless=new") # 新版无头模式更接近真实浏览器 options.add_argument("--no-sandbox") options.add_argument('--disable-blink-features=AutomationControlled') options.add_argument("--disable-extensions") options.add_argument("--disable-gpu") options.add_argument("--disable-dev-shm-usage") # 模拟真实浏览器UA options.add_argument("user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") options.page_load_strategy = 'eager' # 提前解析页面,节省加载时间 # 全局复用浏览器实例 service = Service("/usr/bin/chromedriver") driver = webdriver.Chrome(service=service, options=options) wait = WebDriverWait(driver, 10) # 显式等待最长10秒 # 存储提取结果和失败URL all_data = [] failed_urls = [] def fetch_and_parse_content_selenium(target_url): retries = 3 for attempt in range(retries): try: driver.get(target_url) # 等待核心元素加载(以公司名称h1为例) wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, 'h1.font-extralight'))) # 随机延迟模拟人类操作 time.sleep(random.uniform(1, 3)) # 用lxml解析,速度更快兼容性更好 soup = BeautifulSoup(driver.page_source, 'lxml') # 提取公司名称 company_name2 = soup.select_one('h1.font-extralight').text.strip() if soup.select_one('h1.font-extralight') else None # 提取Batch,增加文本校验避免匹配错误元素 batch = None batch_elem = soup.select_one('div.flex.flex-row.items-center span') if batch_elem and "Batch" in batch_elem.text: batch = batch_elem.text.strip() # 提取Acquisition信息,增加文本校验 acquisition = None acquisition_elem = soup.select_one('div.flex.flex-row.items-center.justify-between span') if acquisition_elem and "Acquired" in acquisition_elem.text: acquisition = acquisition_elem.text.strip() # 提取企业官网URL,转义CSS选择器中的冒号 company_url = None url_elem = soup.select_one('div.inline-block.group-hover\\:underline a') if url_elem: company_url = url_elem.get('href').strip() title = soup.select_one('.prose .text-xl').text.strip() if soup.select_one('.prose .text-xl') else None description = soup.select_one('p.whitespace-pre-line').text.strip() if soup.select_one('p.whitespace-pre-line') else None # 提取Founded/Team Size/Location,遍历匹配文本标签 founded_date = None team_size = None location = None info_rows = soup.select('div.flex.flex-row.justify-between') for row in info_rows: spans = row.find_all('span') if len(spans) >= 2: label = spans[0].text.strip() value = spans[1].text.strip() if label == "Founded:": founded_date = value elif label == "Team Size:": team_size = value elif label == "Location:": location = value # 统计创始人卡片数量,用部分class匹配避免全class匹配失败 founders_count = len(soup.find_all('div', class_=lambda x: x and 'shrink-0' in x and 'bg-[#FDFDF7]' in x)) return { 'company_name2': company_name2, 'batch': batch, 'acquisition': acquisition, 'url': company_url, 'title': title, 'description': description, 'founded_date': founded_date, 'team_size': team_size, 'location': location, 'founders_count': founders_count } except Exception as e: print(f"第{attempt+1}次尝试失败({target_url}):{str(e)}") if attempt == retries - 1: failed_urls.append(target_url) return None # 重试前随机延迟 time.sleep(random.uniform(2, 4)) # 遍历处理所有URL for index, row in df.iterrows(): target_url = row['href'] print(f"处理进度:{index+1}/{len(df)} - {target_url}") data = fetch_and_parse_content_selenium(target_url) if data: all_data.append(data) # 关闭浏览器 driver.quit() # 拼接结果到原DataFrame result_df = pd.DataFrame(all_data) df = pd.concat([df.reset_index(drop=True), result_df.reset_index(drop=True)], axis=1) # 输出结果 print(df) print("\n失败的URL列表:") for failed_url in failed_urls: print(failed_url)
额外注意事项
- 安装依赖:执行
pip install lxml替换默认解析器,提升解析效率和兼容性 - 选择器验证:在浏览器开发者工具中用
document.querySelector("选择器")验证选择器是否唯一匹配目标元素 - 强反爬应对:若网站反爬严格,可替换为
undetected-chromedriver,或添加代理IP、Cookie池
内容的提问来源于stack exchange,提问作者Nick
相关产品推荐
相关产品推荐

