基于LLM的自定义PDF问答项目报错:向量存储加载失败
问题:查询PDF时出现ValueError:找不到vector_store.json
我正在开发一个仅基于自定义PDF进行问答的LLM项目,已完成核心流程代码,但查询PDF时反复遇到以下错误:
ValueError: No existing llama_index.vector_stores.simple found at ./ChromaDb\vector_store.json, skipping load
相关代码
主流程代码
import os, re from grpc import ServicerContext import vectordb from langchain import OpenAI from llama_index import GPTTreeIndex, SimpleDirectoryReader, LLMPredictor,GPTVectorStoreIndex,PromptHelper, VectorStoreIndex from llama_index import LangchainEmbedding, ServiceContext, Prompt from llama_index import StorageContext, load_index_from_storage from langchain.embeddings import OpenAIEmbeddings from langchain.llms import AzureOpenAI import chromadb from llama_index.vector_stores import ChromaVectorStore from dotenv import load_dotenv load_dotenv() persist_directory = './ChromaDb' deployment_name = "text-davinci-003" # Create LLM via Azure OpenAI Service llm = AzureOpenAI(deployment_name=deployment_name) llm_predictor = LLMPredictor(llm=llm) llm_predictor = LLMPredictor(llm = llm_predictor) embedding_llm = LangchainEmbedding(OpenAIEmbeddings()) # Define prompt helper max_input_size = 3000 num_output = 256 chunk_size_limit = 1000 # token window size per document max_chunk_overlap = 20 # overlap for each token fragment prompt_helper = PromptHelper(max_input_size=max_input_size, num_output=num_output, max_chunk_overlap=max_chunk_overlap, chunk_size_limit=chunk_size_limit) def regenrate_tokens(): deployment_name = "text-davinci-003" # loading text data file. documents = SimpleDirectoryReader('./static/upload/').load_data() service_context = ServiceContext.from_defaults(llm_predictor=llm_predictor, embed_model=embedding_llm, prompt_helper=prompt_helper) vector=vectordb.createdb3(documents,embedding_llm,persist_directory,service_context) vector.storage_context.persist(persist_dir= persist_directory) return('Token regenrated, you can ask the questions.') def query__from_knowledge_base(question): if(question == 'regenerate tokens'): return(regenrate_tokens()) # try loading storage_context = StorageContext.from_defaults(persist_dir= persist_directory) # service_context = ServiceContext.from_defaults(llm_predictor=llm_predictor, embed_model=embedding_llm, prompt_helper=prompt_helper) # # load index index = load_index_from_storage(storage_context) # define custom Prompt TEMPLATE_STR = """Create a final answer to the given questions using the provided document excerpts(in no particular order) as references. ALWAYS include a "SOURCES" section in your answer including only the minimal set of sources needed to answer the question. Always include the Source Preview of source. If answer has step in document please response in step. If you are unable to answer the question, simply state that you do not know. Do not attempt to fabricate an answer and leave the SOURCES section empty. "--------------------- " "{context_str}" "\n--------------------- " "Given this information, please answer the question: {query_str}\n" """ QA_TEMPLATE = Prompt(TEMPLATE_STR) query_engine = index.as_query_engine(text_qa_template=QA_TEMPLATE) response = query_engine.query(question) response = str(response) response = re.sub(r'Answer:', '', response) response = response.strip() return(response) #print(regenrate_tokens()) #print(query__from_knowledge_base('Enabling online archive for the user’s mailbox.'))
PDF加载代码
def loadFiles(): loader = DirectoryLoader('./static/upload/', glob="./*.pdf", loader_cls=PyPDFLoader) documents = loader.load() text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=80) texts = text_splitter.split_documents(documents) return texts
向量数据库创建代码
def createdb3(documents,embeddings,persist_directory,service_context): chroma_client = chromadb.Client(Settings( chroma_db_impl="duckdb+parquet", persist_directory= persist_directory)) # create a collection chroma_collection = chroma_client.get_or_create_collection("chromaVectorStore",embedding_function=embeddings) # https://docs.trychroma.com/api-reference print(chroma_collection.count()) vector_store = ChromaVectorStore(chroma_collection) storage_context = StorageContext.from_defaults(vector_store=vector_store) index = GPTVectorStoreIndex.from_documents(documents, storage_context=storage_context, service_context=service_context) print(chroma_collection.count()) print(chroma_collection.get()['documents']) print(chroma_collection.get()['metadatas']) # index.storage_context.persist() return index
解决方法
错误原因
报错本质是:加载StorageContext时,llama-index默认尝试加载SimpleVectorStore的配置文件vector_store.json,但我们实际使用的是ChromaVectorStore,该类型不会生成这个文件,导致加载逻辑不匹配。
具体修复步骤
修复
createdb3函数的缩进错误并完善持久化逻辑
原代码中chroma_collection = ...一行缩进错误,同时补充Chroma数据持久化代码:def createdb3(documents,embeddings,persist_directory,service_context): from chromadb.config import Settings # 导入缺失的Settings类 chroma_client = chromadb.Client(Settings( chroma_db_impl="duckdb+parquet", persist_directory= persist_directory )) # 创建/获取Chroma集合 chroma_collection = chroma_client.get_or_create_collection("chromaVectorStore", embedding_function=embeddings) print(chroma_collection.count()) vector_store = ChromaVectorStore(chroma_collection) storage_context = StorageContext.from_defaults(vector_store=vector_store) index = GPTVectorStoreIndex.from_documents(documents, storage_context=storage_context, service_context=service_context) print(chroma_collection.count()) print(chroma_collection.get()['documents']) print(chroma_collection.get()['metadatas']) # 显式持久化Chroma数据 chroma_client.persist() return index修改查询函数的StorageContext加载逻辑
查询时需要重新初始化Chroma客户端和ChromaVectorStore,再关联到StorageContext,而不是直接用from_defaults。修改后的query__from_knowledge_base函数:def query__from_knowledge_base(question): if(question == 'regenerate tokens'): return(regenrate_tokens()) # 重新初始化Chroma客户端和VectorStore from chromadb.config import Settings chroma_client = chromadb.Client(Settings( chroma_db_impl="duckdb+parquet", persist_directory= persist_directory )) chroma_collection = chroma_client.get_or_create_collection("chromaVectorStore", embedding_function=embedding_llm) vector_store = ChromaVectorStore(chroma_collection) # 创建包含ChromaVectorStore的StorageContext storage_context = StorageContext.from_defaults(vector_store=vector_store, persist_dir=persist_directory) # 加载索引时必须传入service_context service_context = ServiceContext.from_defaults(llm_predictor=llm_predictor, embed_model=embedding_llm, prompt_helper=prompt_helper) index = load_index_from_storage(storage_context, service_context=service_context) # 自定义Prompt(修正格式问题) TEMPLATE_STR = """Create a final answer to the given questions using the provided document excerpts(in no particular order) as references. ALWAYS include a "SOURCES" section in your answer including only the minimal set of sources needed to answer the question. Always include the Source Preview of source. If answer has step in document please response in step. If you are unable to answer the question, simply state that you do not know. Do not attempt to fabricate an answer and leave the SOURCES section empty. --------------------- {context_str} --------------------- Given this information, please answer the question: {query_str} """ QA_TEMPLATE = Prompt(TEMPLATE_STR) query_engine = index.as_query_engine(text_qa_template=QA_TEMPLATE) response = query_engine.query(question) response = str(response).replace('Answer:', '').strip() return(response)额外注意事项
- 确保
persist_directory路径在创建和查询时完全一致,避免跨系统路径格式差异问题 - 运行
regenerate tokens后,检查./ChromaDb目录下是否生成Chroma持久化文件(如chroma.sqlite3、parquet文件夹等) - 确认llama-index和chromadb版本兼容,建议使用最新稳定版
- 确保
内容的提问来源于stack exchange,提问作者Pmd
相关产品推荐
相关产品推荐

