如何使用Python实现MHT/MHTML文件到PPTX格式的快速转换
MHT/MHTML转PPTX Python实现方案
依赖安装
首先安装需要的第三方库:
pip install mht2html beautifulsoup4 python-pptx
实现逻辑
- 第一步:解析MHT文件,提取出HTML文本和关联的图片、样式等本地资源
- 第二步:解析HTML内容,按段落、标题、图片等元素拆分内容块
- 第三步:每个内容块对应生成PPT页面,把文本、图片等元素写入PPTX对应位置
完整代码
from mht2html import MHT2Html from bs4 import BeautifulSoup from pptx import Presentation from pptx.util import Inches, Pt import os import shutil def mht_to_pptx(mht_path, output_pptx_path): # 临时目录存放解析出来的HTML和资源 temp_dir = "./mht_temp" os.makedirs(temp_dir, exist_ok=True) # 1. 解析MHT为HTML mht_parser = MHT2Html(mht_path) html_path = mht_parser.parse(temp_dir) # 2. 读取HTML内容解析 with open(html_path, 'r', encoding='utf-8', errors='ignore') as f: html_content = f.read() soup = BeautifulSoup(html_content, 'html.parser') # 3. 初始化PPT prs = Presentation() # 标题+内容版式 content_layout = prs.slide_layouts[1] current_title = "" current_content = [] for elem in soup.find_all(['h1', 'h2', 'h3', 'p', 'img']): # 处理标题 if elem.name in ['h1', 'h2', 'h3']: # 已有内容先生成上一页 if current_title or current_content: slide = prs.slides.add_slide(content_layout) slide.shapes.title.text = current_title if current_title else "无标题" tf = slide.placeholders[1].text_frame for line in current_content: p = tf.add_paragraph() p.text = line p.font.size = Pt(14) current_content = [] current_title = elem.get_text(strip=True) # 处理普通段落 elif elem.name == 'p': text = elem.get_text(strip=True) if text: current_content.append(text) # 处理图片,单独生成一页 elif elem.name == 'img': img_path = os.path.join(temp_dir, elem['src']) if os.path.exists(img_path): # 先保存之前的文本内容 if current_title or current_content: slide = prs.slides.add_slide(content_layout) slide.shapes.title.text = current_title if current_title else "无标题" tf = slide.placeholders[1].text_frame for line in current_content: p = tf.add_paragraph() p.text = line p.font.size = Pt(14) current_title = "" current_content = [] # 生成图片页 slide = prs.slides.add_slide(prs.slide_layouts[6]) # 空白版式 left = top = Inches(1) slide.shapes.add_picture(img_path, left, top, height=Inches(5.5)) # 处理最后剩余的内容 if current_title or current_content: slide = prs.slides.add_slide(content_layout) slide.shapes.title.text = current_title if current_title else "无标题" tf = slide.placeholders[1].text_frame for line in current_content: p = tf.add_paragraph() p.text = line p.font.size = Pt(14) # 保存PPT prs.save(output_pptx_path) # 清理临时文件 shutil.rmtree(temp_dir) # 调用示例 if __name__ == "__main__": mht_to_pptx("你的输入文件.mht", "输出文件.pptx")
注意事项
- 代码默认按HTML的标题、段落、图片拆分生成独立PPT页,可根据自身需求调整元素拆分逻辑
- 如果MHT里包含表格、复杂样式,需要额外添加解析逻辑适配,
python-pptx原生支持表格生成,可自行扩展 - 若出现编码报错,可根据MHT文件的实际编码调整
open方法里的encoding参数
内容的提问来源于stack exchange,提问作者t2dajay
相关产品推荐
相关产品推荐

