如何用Python实现网站内容与HTML代码实时监控及变动检测
网站变化监控代码实现
完整可运行代码
import requests from bs4 import BeautifulSoup import time import difflib import os # 不需要邮件上报可以注释下面两个导入 import smtplib from email.mime.text import MIMEText # ========== 手动配置项 直接修改这里的参数即可 ========== URL = "https://www.uetmardan.edu.pk/uetm/" # 爬取间隔 单位:秒 示例300为5分钟爬一次 CRAWL_INTERVAL = 300 # 历史内容存储路径 HISTORY_FILE = "last_crawl_content.txt" # True为爬取完整HTML代码 False为只爬取页面文本 FETCH_FULL_HTML = False # 邮件上报配置 不需要开启就把enable设为False SMTP_CONFIG = { "enable": False, "server": "smtp.xxx.com", "port": 465, "user": "你的邮箱地址", "password": "邮箱授权码", "to_addr": "接收通知的邮箱地址" } def get_page_content(): """获取目标页面内容""" try: # 加请求头模拟浏览器访问 避免被站点拦截 headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36" } resp = requests.get(URL, headers=headers, timeout=10) resp.raise_for_status() resp.encoding = resp.apparent_encoding soup = BeautifulSoup(resp.content, 'html5lib') if FETCH_FULL_HTML: return soup.prettify() else: # 过滤无意义的空白字符 避免误判变动 return '\n'.join([line.strip() for line in soup.get_text().splitlines() if line.strip()]) except Exception as e: print(f"爬取出错:{str(e)}") return None def compare_content(old_content, new_content): """对比两次爬取内容的差异 有差异返回差异内容 无差异返回None""" diff = difflib.unified_diff( old_content.splitlines(), new_content.splitlines(), fromfile='上次爬取内容', tofile='本次爬取内容', lineterm='' ) diff_list = list(diff) return '\n'.join(diff_list) if diff_list else None def notify_change(diff_content): """变动上报逻辑""" print("="*60) print("检测到页面内容发生变动!变动详情:") print(diff_content) print("="*60) # 邮件上报逻辑 if SMTP_CONFIG["enable"]: try: msg = MIMEText(f"监控页面 {URL} 发生变动,差异如下:\n{diff_content}", 'plain', 'utf-8') msg['Subject'] = f"网站变动监控通知:{URL}" msg['From'] = SMTP_CONFIG["user"] msg['To'] = SMTP_CONFIG["to_addr"] with smtplib.SMTP_SSL(SMTP_CONFIG["server"], SMTP_CONFIG["port"]) as server: server.login(SMTP_CONFIG["user"], SMTP_CONFIG["password"]) server.sendmail(SMTP_CONFIG["user"], SMTP_CONFIG["to_addr"], msg.as_string()) print("变动通知邮件已发送") except Exception as e: print(f"邮件发送失败:{str(e)}") def main(): # 首次运行初始化基准内容 if not os.path.exists(HISTORY_FILE): print("首次运行,初始化基准内容...") initial_content = get_page_content() if initial_content: with open(HISTORY_FILE, 'w', encoding='utf-8') as f: f.write(initial_content) print("基准内容存储完成,开始监控...") else: print("首次爬取失败,请检查网络或URL配置") return # 循环监控逻辑 while True: time.sleep(CRAWL_INTERVAL) print(f"\n{time.strftime('%Y-%m-%d %H:%M:%S')} 开始新一轮爬取...") # 读取上次存储的内容 with open(HISTORY_FILE, 'r', encoding='utf-8') as f: last_content = f.read() # 获取新内容 new_content = get_page_content() if not new_content: continue # 对比差异 diff = compare_content(last_content, new_content) if diff: notify_change(diff) # 更新存储的内容为最新版本 with open(HISTORY_FILE, 'w', encoding='utf-8') as f: f.write(new_content) else: print("内容无变动") if __name__ == "__main__": main()
功能说明
- 自定义配置:所有可调参数统一放在开头配置块,无需修改业务逻辑即可调整爬取规则
- 内容持久化:首次运行自动存储基准内容,每次检测到变动自动更新本地存储的最新内容
- 定时爬取:通过
time.sleep实现固定间隔自动爬取,资源占用低 - 差异展示:用
difflib生成标准化差异结果,清晰标注新增、删除的内容 - 多渠道上报:默认控制台打印变动,可配置开启邮件上报
- 异常兼容:添加浏览器请求头避免被站点拦截,异常捕获避免单次爬取失败导致程序崩溃,过滤无意义空白字符减少误报
使用说明
- 先安装依赖包:
pip install requests beautifulsoup4 html5lib
- 修改开头配置项的参数,比如爬取间隔、是否抓取完整HTML、邮件配置等
- 直接运行代码即可,首次运行会自动初始化基准内容,后续按设定间隔自动爬取对比
内容的提问来源于stack exchange,提问作者Roman Khattak
相关产品推荐
相关产品推荐

