You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何通过ChatGPT开发API增量构建索引以降低成本?

实现Llama Index增量式索引构建(仅为新增数据付费)

你当前的实现仍在全量重建索引——每次新增数据时都会加载所有旧文档,合并后重新生成整个索引,这会导致你为全部文档重复支付embedding生成费用,完全没达到增量更新的目的。

Llama Index原生支持向现有索引直接添加文档,无需全量重建,只需要为新增数据生成embedding,以此降低成本。以下是修改后的完整实现:

import hashlib

from llama_index import StorageContext, load_index_from_storage, GPTVectorStoreIndex, LLMPredictor, PromptHelper
from langchain import OpenAI
from typing import List
import gradio as gr
import os

os.environ["OPENAI_API_KEY"] = 'xxxxxxxx'

class Document:
    def __init__(self,
                 text,
                 doc_id,
                 metadata=None,
                 extra_info_str: str = "",
                 embedding: List[float] = None,
                 extra_info=None):
        self.text = text
        self.doc_id = doc_id
        self.metadata = metadata if metadata is not None else {}
        self.extra_info_str = extra_info_str
        self.extra_info = extra_info
        self.embedding = embedding

    def get_doc_id(self):
        return self.doc_id

    def get_doc_hash(self):
        return hashlib.md5(self.text.encode('utf-8')).hexdigest()

    def get_text(self):
        return self.text


def construct_index(file_path, checkpoint_file):
    max_input_size = 4096
    num_outputs = 512
    max_chunk_overlap = 20
    chunk_size_limit = 600

    prompt_helper = PromptHelper(max_input_size, num_outputs, max_chunk_overlap, chunk_size_limit=chunk_size_limit)
    llm_predictor = LLMPredictor(llm=OpenAI(temperature=0.7, model_name="text-davinci-003", max_tokens=num_outputs))
    persist_dir = "/Media/Disk1/sandbox/ml/chatgpt/index_storage/"

    # 加载检查点,记录已处理的行数
    checkpoint = 0
    if os.path.exists(checkpoint_file):
        with open(checkpoint_file, "r") as f:
            checkpoint = int(f.read().strip())

    # 读取本次新增的数据
    with open(file_path, "r") as f:
        new_entries = f.readlines()[checkpoint:]
        if not new_entries:
            print("无新增数据,无需更新索引")
            return

        # 合并新增内容并生成唯一文档ID(避免重复添加)
        concatenated_text = ''.join(new_entries)
        doc_id = hashlib.md5(concatenated_text.encode('utf-8')).hexdigest()
        new_document = Document(text=concatenated_text, doc_id=doc_id)

    # 处理索引更新逻辑
    if os.path.exists(persist_dir) and len(os.listdir(persist_dir)) > 0:
        # 加载已存在的索引
        storage_context = StorageContext.from_defaults(persist_dir=persist_dir)
        existing_index = load_index_from_storage(
            storage_context,
            llm_predictor=llm_predictor,
            prompt_helper=prompt_helper
        )
        
        # 直接向现有索引插入新文档(仅为新文档生成embedding)
        existing_index.insert(new_document)
        # 持久化更新后的索引
        existing_index.storage_context.persist(persist_dir=persist_dir)
        updated_index = existing_index
        print("新增文档已成功添加到现有索引")
    else:
        # 首次创建索引
        updated_index = GPTVectorStoreIndex.from_documents(
            [new_document],
            llm_predictor=llm_predictor,
            prompt_helper=prompt_helper
        )
        updated_index.storage_context.persist(persist_dir=persist_dir)
        print("首次索引创建完成")

    # 更新检查点,记录已处理的总行数
    with open(checkpoint_file, "w") as f:
        f.write(str(checkpoint + len(new_entries)))

    return updated_index


def chatbot(input_text):
    persist_dir = "/Media/Disk1/sandbox/ml/chatgpt/index_storage/"
    # 加载最新索引
    storage_context = StorageContext.from_defaults(persist_dir=persist_dir)
    read_index = load_index_from_storage(storage_context)
    query_engine = read_index.as_query_engine()
    response = query_engine.query(input_text)
    return response.response


checkpoint_path = "checkpoint.txt"
index = construct_index("docs/test.txt", checkpoint_path)
iface = gr.Interface(fn=chatbot,
                     inputs=gr.components.Textbox(lines=7, label="Enter your text"),
                     outputs="text",
                     title="My AI Chatbot")

iface.launch(share=True)

关键改动说明

  • 取消全量重建:不再遍历加载所有旧文档,而是直接调用existing_index.insert(new_document),Llama Index会自动处理新文档的embedding生成和索引合并,仅产生新增数据的API费用。
  • 唯一文档ID:用新增文本的MD5哈希作为doc_id,确保相同内容不会被重复添加,避免不必要的费用浪费。
  • 简化逻辑:删除了旧文档合并步骤,减少内存占用,提升代码运行效率。

内容的提问来源于stack exchange,提问作者Yash Tandon

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.21 22:45:02