基于LangChain拆分本地HTML文件并保存Chunk及元数据的问题
解决方案
一、LangChain 优化方案
1. 解决乱码问题
UnstructuredHTMLLoader 默认编码识别可能出错,先检测文件实际编码再指定加载:
import chardet from langchain.document_loaders import UnstructuredHTMLLoader def load_html_with_correct_encoding(file_path): # 检测文件编码 with open(file_path, 'rb') as f: result = chardet.detect(f.read()) # 指定编码加载 loader = UnstructuredHTMLLoader(file_path, encoding=result['encoding']) docs = loader.load() return docs
2. 确保标题与段落不分离的拆分逻辑
放弃HTMLHeaderTextSplitter,改用先提取页面标题作为元数据,再用RecursiveCharacterTextSplitter拆分,拆分时将标题嵌入每个chunk的开头:
from langchain.text_splitter import RecursiveCharacterTextSplitter from bs4 import BeautifulSoup def extract_page_title(file_path): with open(file_path, 'r', encoding='utf-8', errors='ignore') as f: soup = BeautifulSoup(f, 'html.parser') # 优先取<title>标签内容,没有则取h1 title = soup.title.string if soup.title else (soup.h1.get_text(strip=True) if soup.h1 else 'Untitled') return title def process_html_file(file_path, chunk_size=20000, chunk_overlap=200): # 加载文件 docs = load_html_with_correct_encoding(file_path) # 提取页面标题 page_title = extract_page_title(file_path) # 初始化拆分器 text_splitter = RecursiveCharacterTextSplitter( chunk_size=chunk_size, chunk_overlap=chunk_overlap, separators=['\n\n', '\n', ' ', ''] ) # 拆分文本,每个chunk开头加上标题 chunks = text_splitter.split_text(docs[0].page_content) # 给每个chunk添加元数据 chunk_docs = [] for chunk in chunks: chunk_docs.append({ 'content': f"【标题】{page_title}\n{chunk}", 'metadata': {'page_title': page_title, 'source': file_path} }) return chunk_docs # 批量处理文件 import os html_dir = '/path/to/your/html/files' all_chunks = [] for filename in os.listdir(html_dir): if filename.endswith('.html'): file_path = os.path.join(html_dir, filename) chunks = process_html_file(file_path) all_chunks.extend(chunks) # 保存到本地(可上传服务器) import json with open('processed_chunks.json', 'w', encoding='utf-8') as f: json.dump(all_chunks, f, ensure_ascii=False, indent=2)
二、非LangChain 高效方案
直接用BeautifulSoup提取结构化内容,自定义拆分逻辑,性能更可控:
import os import json import chardet from bs4 import BeautifulSoup def process_html_non_langchain(file_path, chunk_size=20000): # 检测编码 with open(file_path, 'rb') as f: result = chardet.detect(f.read()) encoding = result['encoding'] or 'utf-8' # 解析HTML with open(file_path, 'r', encoding=encoding, errors='ignore') as f: soup = BeautifulSoup(f, 'html.parser') # 提取标题 page_title = soup.title.string if soup.title else (soup.h1.get_text(strip=True) if soup.h1 else 'Untitled') # 提取结构化内容:标题+对应段落(按h2/h3等标题分组,确保每组标题和内容在一起) content_blocks = [] current_title = page_title current_content = [] # 遍历所有块级元素 for element in soup.find_all(['h1', 'h2', 'h3', 'p', 'div']): if element.name.startswith('h'): # 遇到新标题,先保存之前的块 if current_content: content_blocks.append(f"【{current_title}】\n{' '.join(current_content)}") current_content = [] current_title = element.get_text(strip=True) else: text = element.get_text(strip=True) if text: current_content.append(text) # 保存最后一个块 if current_content: content_blocks.append(f"【{current_title}】\n{' '.join(current_content)}") # 拆分块为符合大小的chunk,确保不拆分单个标题-内容块 chunks = [] for block in content_blocks: if len(block) <= chunk_size: chunks.append(block) else: # 大内容块拆分,保留标题在每个子chunk开头 title_part = block.split('\n')[0] + '\n' content_part = block.split('\n', 1)[1] # 按chunk_size拆分内容部分,每个子chunk开头加标题 start = 0 while start < len(content_part): end = start + chunk_size - len(title_part) # 找最近的换行或空格拆分 if end < len(content_part): end = content_part.rfind(' ', start, end) or end sub_chunk = title_part + content_part[start:end].strip() chunks.append(sub_chunk) start = end # 添加元数据 chunk_docs = [] for chunk in chunks: chunk_docs.append({ 'content': chunk, 'metadata': {'page_title': page_title, 'source': file_path} }) return chunk_docs # 批量处理 html_dir = '/path/to/your/html/files' all_chunks = [] for filename in os.listdir(html_dir): if filename.endswith('.html'): file_path = os.path.join(html_dir, filename) chunks = process_html_non_langchain(file_path) all_chunks.extend(chunks) # 保存 with open('processed_chunks_non_langchain.json', 'w', encoding='utf-8') as f: json.dump(all_chunks, f, ensure_ascii=False, indent=2)
关键说明
- 乱码解决:通过
chardet检测文件实际编码,避免加载时编码不匹配导致乱码。 - 标题与段落不分离:无论是LangChain还是非LangChain方案,都优先将标题与对应内容绑定为一个块,拆分时要么整体保留,要么拆分内容部分但保留标题在每个子chunk开头。
- 元数据保存:每个chunk都附带页面标题和文件路径元数据,满足训练需求。
内容的提问来源于stack exchange,提问作者R_Student
相关产品推荐
相关产品推荐

