如何用Python结合Selenium/bs4定时爬取数据并更新至其他网站?
最优实现方案
一、工具选型
- 静态页面(内容直接在HTML源码中):优先用
BeautifulSoup4 + requests,轻量、速度快、资源占用低,适配高频定时任务场景。 - 动态渲染页面(内容由JS加载生成):用Selenium(或更现代的Playwright),能模拟浏览器执行JS渲染完整页面,但资源占用较高,仅在必须处理动态内容时使用。
二、核心流程拆解
- 可靠定时触发:用APScheduler实现任务调度,比
time.sleep()更灵活,支持异常重启、任务持久化,适合长期运行的定时任务。 - 更新检测逻辑:记录上次抓取的内容特征(如哈希值、内容ID列表、最后更新时间戳),对比本次抓取结果,仅处理新增或变化的内容,避免重复操作。
- 内容提取与格式化:从页面中提取目标字段(标题、正文、链接等),转换成目标网站要求的格式(JSON片段、HTML文本等)。
- 目标网站发布:优先调用目标网站的官方API;若无API,用
requests模拟表单提交(尽量避免用Selenium模拟人工操作,效率低且易被识别)。
三、代码示例
1. 静态页面场景(bs4+APScheduler)
import hashlib import json import requests from bs4 import BeautifulSoup from apscheduler.schedulers.blocking import BlockingScheduler # 持久化存储上次哈希,避免程序重启丢失状态 HASH_FILE = "last_content_hash.json" def load_last_hash(): try: with open(HASH_FILE, "r") as f: return json.load(f)["hash"] except (FileNotFoundError, KeyError): return "" def save_last_hash(hash_str): with open(HASH_FILE, "w") as f: json.dump({"hash": hash_str}, f) def fetch_and_publish(): last_hash = load_last_hash() # 1. 抓取目标页面 target_url = "https://目标网站地址" headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"} try: response = requests.get(target_url, headers=headers, timeout=10) response.raise_for_status() except Exception as e: print(f"抓取失败:{str(e)}") return # 2. 提取核心内容(示例:提取文章标题与链接) soup = BeautifulSoup(response.text, "html.parser") articles = soup.select(".article-list .item") current_content = "\n".join([f"{a.h2.text.strip()}|{a.a['href']}" for a in articles]) # 3. 检测更新 current_hash = hashlib.md5(current_content.encode()).hexdigest() if current_hash == last_hash: print("无更新,跳过发布") return save_last_hash(current_hash) # 4. 自定义格式整理 formatted_content = "\n\n".join([f"【新内容】{a.h2.text.strip()}\n详情链接:{a.a['href']}" for a in articles]) # 5. 发布到目标网站(示例:POST请求提交) publish_url = "https://发布目标地址/api/submit" publish_data = {"content": formatted_content, "auth_token": "你的授权令牌"} try: publish_res = requests.post(publish_url, json=publish_data, timeout=10) publish_res.raise_for_status() print("发布成功") except Exception as e: print(f"发布失败:{str(e)}") # 每小时执行一次任务 scheduler = BlockingScheduler() scheduler.add_job(fetch_and_publish, "interval", hours=1) scheduler.start()
2. 动态页面场景(Selenium+APScheduler)
import hashlib import json from selenium import webdriver from selenium.webdriver.common.by import By from apscheduler.schedulers.blocking import BlockingScheduler HASH_FILE = "last_dynamic_hash.json" def load_last_hash(): try: with open(HASH_FILE, "r") as f: return json.load(f)["hash"] except (FileNotFoundError, KeyError): return "" def save_last_hash(hash_str): with open(HASH_FILE, "w") as f: json.dump({"hash": hash_str}, f) def fetch_and_publish(): last_hash = load_last_hash() # 初始化无头浏览器,减少资源占用 options = webdriver.ChromeOptions() options.add_argument("--headless=new") options.add_argument("--disable-gpu") options.add_argument("--no-sandbox") driver = webdriver.Chrome(options=options) try: driver.get("https://动态目标网站地址") driver.implicitly_wait(10) # 等待JS加载完成 # 提取动态渲染的内容 articles = driver.find_elements(By.CSS_SELECTOR, ".dynamic-article-item") current_content = "\n".join([f"{a.find_element(By.TAG_NAME, 'h3').text}|{a.get_attribute('data-url')}" for a in articles]) # 更新检测 current_hash = hashlib.md5(current_content.encode()).hexdigest() if current_hash == last_hash: print("无动态更新,跳过") return save_last_hash(current_hash) # 格式化与发布(同静态场景,略) formatted_content = "\n\n".join([f"【动态新内容】{a.find_element(By.TAG_NAME, 'h3').text}\n链接:{a.get_attribute('data-url')}" for a in articles]) # 发布代码... print("动态内容发布成功") except Exception as e: print(f"动态抓取/发布失败:{str(e)}") finally: driver.quit() scheduler = BlockingScheduler() scheduler.add_job(fetch_and_publish, "interval", hours=1) scheduler.start()
四、关键优化建议
- 异常兜底:给网络请求、元素提取、发布步骤添加
try-except捕获异常,避免单个任务失败导致整个调度崩溃。 - 反爬规避:静态场景轮换User-Agent,动态场景用
selenium-stealth插件隐藏自动化特征;必要时搭配代理IP池。 - 日志记录:用Python内置
logging模块记录任务执行时间、抓取结果、异常详情,方便后续排查问题。 - 后台运行:将脚本部署为后台进程(用
nohup或systemd服务),确保任务持续运行。
内容的提问来源于stack exchange,提问作者Igetis
相关产品推荐
相关产品推荐

