You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于LangChain拆分本地HTML文件并保存Chunk及元数据的问题

解决方案

一、LangChain 优化方案

1. 解决乱码问题

UnstructuredHTMLLoader 默认编码识别可能出错,先检测文件实际编码再指定加载:

import chardet
from langchain.document_loaders import UnstructuredHTMLLoader

def load_html_with_correct_encoding(file_path):
    # 检测文件编码
    with open(file_path, 'rb') as f:
        result = chardet.detect(f.read())
    # 指定编码加载
    loader = UnstructuredHTMLLoader(file_path, encoding=result['encoding'])
    docs = loader.load()
    return docs

2. 确保标题与段落不分离的拆分逻辑

放弃HTMLHeaderTextSplitter,改用先提取页面标题作为元数据,再用RecursiveCharacterTextSplitter拆分,拆分时将标题嵌入每个chunk的开头:

from langchain.text_splitter import RecursiveCharacterTextSplitter
from bs4 import BeautifulSoup

def extract_page_title(file_path):
    with open(file_path, 'r', encoding='utf-8', errors='ignore') as f:
        soup = BeautifulSoup(f, 'html.parser')
    # 优先取<title>标签内容,没有则取h1
    title = soup.title.string if soup.title else (soup.h1.get_text(strip=True) if soup.h1 else 'Untitled')
    return title

def process_html_file(file_path, chunk_size=20000, chunk_overlap=200):
    # 加载文件
    docs = load_html_with_correct_encoding(file_path)
    # 提取页面标题
    page_title = extract_page_title(file_path)
    # 初始化拆分器
    text_splitter = RecursiveCharacterTextSplitter(
        chunk_size=chunk_size,
        chunk_overlap=chunk_overlap,
        separators=['\n\n', '\n', ' ', '']
    )
    # 拆分文本,每个chunk开头加上标题
    chunks = text_splitter.split_text(docs[0].page_content)
    # 给每个chunk添加元数据
    chunk_docs = []
    for chunk in chunks:
        chunk_docs.append({
            'content': f"【标题】{page_title}\n{chunk}",
            'metadata': {'page_title': page_title, 'source': file_path}
        })
    return chunk_docs

# 批量处理文件
import os

html_dir = '/path/to/your/html/files'
all_chunks = []
for filename in os.listdir(html_dir):
    if filename.endswith('.html'):
        file_path = os.path.join(html_dir, filename)
        chunks = process_html_file(file_path)
        all_chunks.extend(chunks)

# 保存到本地(可上传服务器)
import json
with open('processed_chunks.json', 'w', encoding='utf-8') as f:
    json.dump(all_chunks, f, ensure_ascii=False, indent=2)

二、非LangChain 高效方案

直接用BeautifulSoup提取结构化内容,自定义拆分逻辑,性能更可控:

import os
import json
import chardet
from bs4 import BeautifulSoup

def process_html_non_langchain(file_path, chunk_size=20000):
    # 检测编码
    with open(file_path, 'rb') as f:
        result = chardet.detect(f.read())
    encoding = result['encoding'] or 'utf-8'
    # 解析HTML
    with open(file_path, 'r', encoding=encoding, errors='ignore') as f:
        soup = BeautifulSoup(f, 'html.parser')
    
    # 提取标题
    page_title = soup.title.string if soup.title else (soup.h1.get_text(strip=True) if soup.h1 else 'Untitled')
    
    # 提取结构化内容:标题+对应段落(按h2/h3等标题分组,确保每组标题和内容在一起)
    content_blocks = []
    current_title = page_title
    current_content = []
    # 遍历所有块级元素
    for element in soup.find_all(['h1', 'h2', 'h3', 'p', 'div']):
        if element.name.startswith('h'):
            # 遇到新标题,先保存之前的块
            if current_content:
                content_blocks.append(f"【{current_title}】\n{' '.join(current_content)}")
                current_content = []
            current_title = element.get_text(strip=True)
        else:
            text = element.get_text(strip=True)
            if text:
                current_content.append(text)
    # 保存最后一个块
    if current_content:
        content_blocks.append(f"【{current_title}】\n{' '.join(current_content)}")
    
    # 拆分块为符合大小的chunk,确保不拆分单个标题-内容块
    chunks = []
    for block in content_blocks:
        if len(block) <= chunk_size:
            chunks.append(block)
        else:
            # 大内容块拆分,保留标题在每个子chunk开头
            title_part = block.split('\n')[0] + '\n'
            content_part = block.split('\n', 1)[1]
            # 按chunk_size拆分内容部分,每个子chunk开头加标题
            start = 0
            while start < len(content_part):
                end = start + chunk_size - len(title_part)
                # 找最近的换行或空格拆分
                if end < len(content_part):
                    end = content_part.rfind(' ', start, end) or end
                sub_chunk = title_part + content_part[start:end].strip()
                chunks.append(sub_chunk)
                start = end
    
    # 添加元数据
    chunk_docs = []
    for chunk in chunks:
        chunk_docs.append({
            'content': chunk,
            'metadata': {'page_title': page_title, 'source': file_path}
        })
    return chunk_docs

# 批量处理
html_dir = '/path/to/your/html/files'
all_chunks = []
for filename in os.listdir(html_dir):
    if filename.endswith('.html'):
        file_path = os.path.join(html_dir, filename)
        chunks = process_html_non_langchain(file_path)
        all_chunks.extend(chunks)

# 保存
with open('processed_chunks_non_langchain.json', 'w', encoding='utf-8') as f:
    json.dump(all_chunks, f, ensure_ascii=False, indent=2)

关键说明

  • 乱码解决:通过chardet检测文件实际编码,避免加载时编码不匹配导致乱码。
  • 标题与段落不分离:无论是LangChain还是非LangChain方案,都优先将标题与对应内容绑定为一个块,拆分时要么整体保留,要么拆分内容部分但保留标题在每个子chunk开头。
  • 元数据保存:每个chunk都附带页面标题和文件路径元数据,满足训练需求。

内容的提问来源于stack exchange,提问作者R_Student

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.23 22:48:13