基于LangChain构建企业文档向量库及多文档LLM摘要的技术咨询
基于LangChain实现多文档扫描与LLM结合的方案
一、替换为HuggingFace Sentence-Transformer Embedding
直接用LangChain的HuggingFaceEmbeddings类替换自研Embedding,无需修改FAISS核心逻辑,只需在构建/加载索引时指定该Embedding:
from langchain.embeddings import HuggingFaceEmbeddings # 初始化Sentence-Transformer Embedding embedding_model = HuggingFaceEmbeddings( model_name="all-MiniLM-L6-v2", # 可替换为目标模型 model_kwargs={"device": "cuda"} # 按需指定运行设备 ) # 重新构建索引示例(若已有索引,可加载后绑定该Embedding) from langchain.vectorstores import FAISS # db = FAISS.from_documents(documents, embedding_model) # db.save_local("faiss_index") # 加载已有索引并绑定新Embedding db = FAISS.load_local("faiss_index", embedding_model)
二、整合自定义Llama训练的LLM
通过继承LangChain的LLM类,封装自定义Llama模型的调用逻辑:
from langchain.llms.base import LLM from typing import Optional, List, Mapping, Any class CustomLlamaLLM(LLM): model_path: str device: str = "cuda" @property def _llm_type(self) -> str: return "custom_llama" def _call( self, prompt: str, stop: Optional[List[str]] = None, run_manager: Optional[Any] = None, ) -> str: # 替换为你的自定义Llama模型推理逻辑 from llama_cpp import Llama llm = Llama(model_path=self.model_path, n_ctx=4096, device=self.device) output = llm(prompt, stop=stop, max_tokens=512) return output["choices"][0]["text"].strip() @property def _identifying_params(self) -> Mapping[str, Any]: return {"model_path": self.model_path, "device": self.device} # 初始化自定义LLM custom_llm = CustomLlamaLLM(model_path="./your-trained-llama-model.gguf")
三、多文档扫描与LLM生成摘要的核心流程
1. 带日期过滤的向量检索
先执行相似检索,再根据文档metadata中的日期字段过滤结果:
def similarity_search_with_date_filter(query, index, start_date=None, end_date=None): # 先获取较多候选结果再过滤 matched_docs = index.similarity_search(query, k=20) filtered_docs = [] for doc in matched_docs: doc_date = doc.metadata.get("date") if not doc_date: continue # 日期比较逻辑按需调整格式 if start_date and doc_date < start_date: continue if end_date and doc_date > end_date: continue filtered_docs.append(doc) sources = [{ "page_content": doc.page_content, "metadata": doc.metadata } for doc in filtered_docs[:5]] # 最终取前5条有效结果 return filtered_docs[:5], sources
2. 多文档摘要生成(两种可选方案)
方案一:Stuff模式(适合短文档集合)
直接拼接所有文档内容,由LLM生成统一摘要:
from langchain.chains.summarize import load_summarize_chain from langchain.prompts import PromptTemplate # 自定义摘要Prompt prompt_template = """请结合用户查询{query},总结以下文档的核心信息: {text} 总结:""" PROMPT = PromptTemplate(template=prompt_template, input_variables=["query", "text"]) # 加载Stuff链 stuff_chain = load_summarize_chain( llm=custom_llm, chain_type="stuff", prompt=PROMPT ) # 生成摘要 def generate_summary(query, docs): return stuff_chain.run(query=query, docs=docs)
方案二:Map-Reduce模式(适合海量长文档)
先对单文档生成子摘要,再合并为最终总摘要:
# Map阶段Prompt(单文档摘要) map_prompt = """总结以下单篇文档的核心内容: {text} 单篇总结:""" MAP_PROMPT = PromptTemplate(template=map_prompt, input_variables=["text"]) # Reduce阶段Prompt(合并子摘要) reduce_prompt = """结合用户查询{query},将以下多个子摘要合并为连贯全面的总总结: {text} 总总结:""" REDUCE_PROMPT = PromptTemplate(template=reduce_prompt, input_variables=["query", "text"]) # 加载Map-Reduce链 map_reduce_chain = load_summarize_chain( llm=custom_llm, chain_type="map_reduce", map_prompt=MAP_PROMPT, combine_prompt=REDUCE_PROMPT ) # 生成摘要 def generate_map_reduce_summary(query, docs): return map_reduce_chain.run(query=query, docs=docs)
3. 完整流程调用示例
# 加载向量库和LLM db = FAISS.load_local("faiss_index", embedding_model) custom_llm = CustomLlamaLLM(model_path="./your-trained-llama-model.gguf") # 用户查询与日期过滤 user_query = "2024年Q1市场调研报告核心结论" start_date = "2024-01-01" end_date = "2024-03-31" matched_docs, sources = similarity_search_with_date_filter(user_query, db, start_date, end_date) # 生成并输出结果 if matched_docs: summary = generate_map_reduce_summary(user_query, matched_docs) print("生成的摘要:", summary) print("来源文档:", sources) else: print("未找到符合条件的文档")
内容的提问来源于stack exchange,提问作者Tanmoy
相关产品推荐
相关产品推荐

