使用Langchain与Python向ChromaDB插入JSON对象失败问题排查
问题:批量导入JSON格式Topic到ChromaDB失败,仅7条成功
我有35个合法JSON格式的Topic(其中34个为唯一值),这些JSON可在MongoDB中正常增删,也能通过json.dumps和json.loads处理,是通过LangChain调用OpenAI并经JSON格式化工具生成的。但尝试导入ChromaDB时,仅7个Topic成功插入,且使用RecursiveJsonSplitter始终报错,却从未收到Chroma的报错信息,测试脚本输出显示有35条记录,但实际数据库中只有7条。
导入ChromaDB的函数
def add_to_chroma(database_name: str, collection_name: str, json_object: json, user_directory: str, state, meta_data: json, doc_ids: list[str] = None): try: db_directory = os.path.join(user_directory, database_name + ".db") embedding_function = SentenceTransformerEmbeddings(model_name="all-MiniLM-L6-v2") # embedding_function = OpenAIEmbeddings(model="text-embedding-3-small") chroma_db = Chroma(persist_directory=db_directory, collection_name=collection_name, embedding_function=embedding_function, collection_metadata={"hnsw:space": "cosine"}) # embed_object = write_object_to_prompt(json_object) embed_object = json_object # text_splitter = RecursiveCharacterTextSplitter(chunk_size=2000, chunk_overlap=100) json_splitter = RecursiveJsonSplitter(max_chunk_size=2000) json_docs = json_splitter.split_json(embed_object, True) meta_list = [] for json_doc in json_docs: meta_list.append(meta_data) docs = json_splitter.create_documents(texts=json_docs, metadatas=meta_list) if doc_ids is None: doc_ids = [str(uuid.uuid4()) for i in range(1, len(docs) + 1)] else: # We look to see if the document exists: result = chroma_db.get(doc_ids) if result is not None and len(result) > 0: # This is an update: state["persistent_logs"].append("Updating " + meta_data["topic_id"] + " in Chroma") chroma_db.update_documents(doc_ids, docs) return doc_ids, state state["persistent_logs"].append("Adding " + meta_data["topic_id"] + " to Chroma") chroma_db.from_documents(docs, embedding_function, ids=doc_ids) except: trace_back = traceback.format_exc() logging.error("An unexpected error occurred attempting to add document to Chroma: " + database_name + ", to the collection: " + collection_name + "\nHere is the document that failed: " + write_object_to_prompt(json_object) + " \nWith the error:\n " + trace_back) state["persistent_logs"].append( "An unexpected error occurred attempting to add document to Chroma: " + database_name + ", to the collection: " + collection_name + "\nHere is the document that failed: " + write_object_to_prompt(json_object) + " \nWith the error:\n " + trace_back) return doc_ids, state
查询ChromaDB记录数的函数
def get_current_count(database_name: str, collection_name: str, user_directory: str) -> int: db_directory = os.path.join(user_directory, database_name + ".db") # embedding_function = SentenceTransformerEmbeddings(model_name="all-MiniLM-L6-v2") embedding_function = OpenAIEmbeddings(model="text-embedding-3-small") # Cosine will keep te similarity scores between zero and one chroma_db = Chroma(persist_directory=db_directory, collection_name=collection_name, embedding_function=embedding_function, collection_metadata={"hnsw:space": "cosine"}) results = chroma_db.get() total_count = 0 if results is not None: total_count = len(results) return total_count
测试脚本
user_directory = "../UserData/user-x" with open("Topics.txt", "r") as f: Topics = json.load(f) print("Total topics: " + str(len(Topics))) state = { "errors": "", "persistent_logs": [], } unique_ids = [] for this_topic in Topics: meta = {"topic_id": this_topic["topic_id"]} doc_ids, state = add_to_chroma("user-x", "conversations", this_topic, user_directory, state, meta) if this_topic["topic_id"] not in unique_ids: unique_ids.append(this_topic["topic_id"]) print("Number of unique Ids: " + str(len(unique_ids))) if len(state["persistent_logs"]) > 0: for log in state["persistent_logs"]: print(log) print("Total Topics in Chroma " + str(get_current_count("user-x", "topics", user_directory)))
已尝试的解决方案
- 切换使用OpenAI的
text-embedding-3-small模型与HuggingFace的all-MiniLM-L6-v2嵌入模型 - 尝试
RecursiveJsonSplitter和RecursiveCharacterTextSplitter两种文本拆分器 - 手动创建ChromaDB的
Document对象 - 分别以纯文本和JSON格式导入数据
导入的模块
import json import uuid from langchain_chroma import Chroma from langchain.docstore.document import Document from langchain_text_splitters import RecursiveCharacterTextSplitter, RecursiveJsonSplitter from langchain_community.embeddings.sentence_transformer import SentenceTransformerEmbeddings from langchain_openai import OpenAIEmbeddings import logging import os import traceback
求可行的解决思路?
内容的提问来源于stack exchange,提问作者Ken Tola
相关产品推荐
相关产品推荐

