You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Langchain与Python向ChromaDB插入JSON对象失败问题排查

问题:批量导入JSON格式Topic到ChromaDB失败,仅7条成功

我有35个合法JSON格式的Topic(其中34个为唯一值),这些JSON可在MongoDB中正常增删,也能通过json.dumps和json.loads处理,是通过LangChain调用OpenAI并经JSON格式化工具生成的。但尝试导入ChromaDB时,仅7个Topic成功插入,且使用RecursiveJsonSplitter始终报错,却从未收到Chroma的报错信息,测试脚本输出显示有35条记录,但实际数据库中只有7条。

导入ChromaDB的函数

def add_to_chroma(database_name: str, collection_name: str, json_object: json, user_directory: str, state, meta_data: json, doc_ids: list[str] = None):
    try:
        db_directory = os.path.join(user_directory, database_name + ".db")
        embedding_function = SentenceTransformerEmbeddings(model_name="all-MiniLM-L6-v2")
        # embedding_function = OpenAIEmbeddings(model="text-embedding-3-small")
        chroma_db = Chroma(persist_directory=db_directory, collection_name=collection_name, embedding_function=embedding_function,
                           collection_metadata={"hnsw:space": "cosine"})
        # embed_object = write_object_to_prompt(json_object)
        embed_object = json_object
        # text_splitter = RecursiveCharacterTextSplitter(chunk_size=2000, chunk_overlap=100)
        json_splitter = RecursiveJsonSplitter(max_chunk_size=2000)
        json_docs = json_splitter.split_json(embed_object, True)
        meta_list = []
        for json_doc in json_docs:
            meta_list.append(meta_data)
        docs = json_splitter.create_documents(texts=json_docs, metadatas=meta_list)

        if doc_ids is None:
            doc_ids = [str(uuid.uuid4()) for i in range(1, len(docs) + 1)]
        else:
            # We look to see if the document exists:
            result = chroma_db.get(doc_ids)
            if result is not None and len(result) > 0:
                # This is an update:
                state["persistent_logs"].append("Updating " + meta_data["topic_id"] + " in Chroma")
                chroma_db.update_documents(doc_ids, docs)
                return doc_ids, state
        state["persistent_logs"].append("Adding " + meta_data["topic_id"] + " to Chroma")
        chroma_db.from_documents(docs, embedding_function, ids=doc_ids)
    except:
        trace_back = traceback.format_exc()
        logging.error("An unexpected error occurred attempting to add document to Chroma: " + database_name + ", to the collection: " + collection_name +
                      "\nHere is the document that failed: " + write_object_to_prompt(json_object) + " \nWith the error:\n " + trace_back)
        state["persistent_logs"].append(
            "An unexpected error occurred attempting to add document to Chroma: " + database_name + ", to the collection: " + collection_name +
            "\nHere is the document that failed: " + write_object_to_prompt(json_object) + " \nWith the error:\n " + trace_back)
    return doc_ids, state

查询ChromaDB记录数的函数

def get_current_count(database_name: str, collection_name: str, user_directory: str) -> int:
    db_directory = os.path.join(user_directory, database_name + ".db")
    # embedding_function = SentenceTransformerEmbeddings(model_name="all-MiniLM-L6-v2")
    embedding_function = OpenAIEmbeddings(model="text-embedding-3-small")
    # Cosine will keep te similarity scores between zero and one
    chroma_db = Chroma(persist_directory=db_directory, collection_name=collection_name, embedding_function=embedding_function,
                       collection_metadata={"hnsw:space": "cosine"})
    results = chroma_db.get()
    total_count = 0
    if results is not None:
        total_count = len(results)
    return total_count

测试脚本

user_directory = "../UserData/user-x"
with open("Topics.txt", "r") as f:
    Topics = json.load(f)
print("Total topics: " + str(len(Topics)))
state = {
        "errors": "",
        "persistent_logs": [],
    }
unique_ids = []
for this_topic in Topics:
    meta = {"topic_id": this_topic["topic_id"]}
    doc_ids, state = add_to_chroma("user-x", "conversations", this_topic, user_directory, state, meta)
    if this_topic["topic_id"] not in unique_ids:
        unique_ids.append(this_topic["topic_id"])
print("Number of unique Ids: " + str(len(unique_ids)))

if len(state["persistent_logs"]) > 0:
    for log in state["persistent_logs"]:
        print(log)
print("Total Topics in Chroma " + str(get_current_count("user-x", "topics", user_directory)))

已尝试的解决方案

  • 切换使用OpenAI的text-embedding-3-small模型与HuggingFace的all-MiniLM-L6-v2嵌入模型
  • 尝试RecursiveJsonSplitter和RecursiveCharacterTextSplitter两种文本拆分器
  • 手动创建ChromaDB的Document对象
  • 分别以纯文本和JSON格式导入数据

导入的模块

import json
import uuid
from langchain_chroma import Chroma
from langchain.docstore.document import Document
from langchain_text_splitters import RecursiveCharacterTextSplitter, RecursiveJsonSplitter
from langchain_community.embeddings.sentence_transformer import SentenceTransformerEmbeddings
from langchain_openai import OpenAIEmbeddings
import logging
import os
import traceback

求可行的解决思路?


内容的提问来源于stack exchange,提问作者Ken Tola

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 19:29:59