使用Python批量将目录下多个HTML文件转换为同名TXT文件的方法
批量HTML转TXT实现方法
你可以用Python标准库os完成目录遍历,复用已有的单文件解析逻辑,批量处理所有html文件,转换后的txt文件会沿用原文件名,仅替换后缀为.txt。
基础版本(仅处理目标目录下的html,不遍历子文件夹)
import os from bs4 import BeautifulSoup # 替换为你存放html文件的实际目录路径 html_dir = "/content/drive/MyDrive/你的html文件目录" for file_name in os.listdir(html_dir): # 筛选后缀为.html的文件,兼容大写后缀的情况 if file_name.lower().endswith(".html"): # 拼接文件完整路径 html_full_path = os.path.join(html_dir, file_name) # 生成对应txt文件的路径 txt_file_name = f"{os.path.splitext(file_name)[0]}.txt" txt_full_path = os.path.join(html_dir, txt_file_name) # 读取html解析纯文本,指定编码避免乱码,with语法自动关闭文件 with open(html_full_path, "r", encoding="utf-8") as f: soup = BeautifulSoup(f.read(), "html.parser") # 写入txt文件 with open(txt_full_path, "w", encoding="utf-8") as f: f.write(soup.get_text()) print(f"已转换:{file_name} -> {txt_file_name}")
可选调整项
- 如果需要递归处理目录下所有子文件夹里的html文件,把遍历逻辑替换为
os.walk实现即可,核心代码如下:
import os from bs4 import BeautifulSoup html_dir = "/content/drive/MyDrive/你的html文件目录" # 递归遍历所有子目录 for root, _, files in os.walk(html_dir): for file_name in files: if file_name.lower().endswith(".html"): html_full_path = os.path.join(root, file_name) txt_file_name = f"{os.path.splitext(file_name)[0]}.txt" txt_full_path = os.path.join(root, txt_file_name) with open(html_full_path, "r", encoding="utf-8") as f: soup = BeautifulSoup(f.read(), "html.parser") with open(txt_full_path, "w", encoding="utf-8") as f: f.write(soup.get_text()) print(f"已转换:{html_full_path} -> {txt_full_path}")
- 如果你的html文件不是utf-8编码,把代码里的
encoding="utf-8"改成对应编码即可,比如gbk、gb2312 - 如果不想把txt存在原目录,只需要修改
txt_full_path的拼接路径,指向你要存放txt的目标目录就行 - 代码里给BeautifulSoup指定了
"html.parser"解析器,不需要额外安装其他解析依赖,和你原来的单文件转换效果完全一致
内容的提问来源于stack exchange,提问作者Asalrayzah
相关产品推荐
相关产品推荐

