You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

LangChain中FAISS向量存储调用delete方法报错的解决办法

问题

尝试通过元数据source过滤删除FAISS向量存储中的数据,调用db.delete(list)时出现错误:

NotImplementedError: delete method must be implemented by subclass.

用户原代码:

db = FAISS.load_local(FAISS_USERGUIDE_INDEX, embeddings)
def store_to_df(store):
    v_dict=store.docstore._dict
    data_rows=[]
    for k in v_dict.keys():
        doc_name=v_dict[k].metadata['source'].split('/')[-1]
        page_number=v_dict[k].metadata['page']+1
        content=v_dict[k].page_content
        data_rows.append({"chunk_id":k,"document":doc_name,"page":page_number,"content":content})
    vector_df=pd.DataFrame(data_rows)
    return vector_df
        
def delete_document(store,document):
    vector_df=store_to_df(store)
    chunks_list=vector_df.loc[vector_df['document']==document]['chunk_id'].tolist()
    store.delete(chunks_list)
delete_document(db,"doc2.pdf")

报错栈:

NotImplementedError                       Traceback (most recent call last)
Cell In[115], line 17
     15     chunks_list=vector_df.loc[vector_df['document']==document]['chunk_id'].tolist()
     16     store.delete(chunks_list)
---> 17 delete_document(db,"doc2.pdf")

Cell In[115], line 16, in delete_document(store, document)
     14 vector_df=store_to_df(store)
     15 chunks_list=vector_df.loc[vector_df['document']==document]['chunk_id'].tolist()
---> 16 store.delete(chunks_list)

File ~\anaconda\anaconda\Lib\site-packages\langchain\vectorstores\base.py:81, in delete(self, ids, **kwargs)
     67     """Delete by vector ID or other criteria.
     68 
     69     Args:
   (...)
     75         False otherwise, None if not implemented.
     76     """
     78     raise NotImplementedError("delete method must be implemented by subclass.")
     80 async def aadd_texts(
---> 81     self,
     82     texts: Iterable[str],
     83     metadatas: Optional[List[dict]] = None,
     84     **kwargs: Any,
     85 ) -> List[str]:
     86     """Run more texts through the embeddings and add to the vectorstore."""
     87     raise NotImplementedError

NotImplementedError: delete method must be implemented by subclass.
解决方案

LangChain早期版本的FAISS向量存储类未实现delete方法,需要直接操作FAISS的内部结构完成删除,步骤如下:

  1. 获取待删除的chunk ID列表(用户原代码已实现此部分)
  2. 从FAISS索引中移除对应向量:通过反向映射找到chunk ID对应的索引位置,批量删除
  3. 清理文档存储和映射关系:删除docstore中的文档条目,更新index_to_docstore_id映射
  4. 持久化修改:将修改后的FAISS库重新保存到本地

修改后的delete_document函数如下:

def delete_document(store, document):
    vector_df = store_to_df(store)
    chunks_list = vector_df.loc[vector_df['document'] == document]['chunk_id'].tolist()
    
    # 反向获取要删除的索引位置
    indices_to_delete = []
    for idx, doc_id in enumerate(store.index_to_docstore_id):
        if doc_id in chunks_list:
            indices_to_delete.append(idx)
    
    if indices_to_delete:
        # 转换为numpy数组并按降序排列(避免删除时索引移位)
        import numpy as np
        indices_to_delete_np = np.array(indices_to_delete, dtype=np.int64)
        indices_to_delete_np = np.sort(indices_to_delete_np)[::-1]
        store.index.remove_ids(indices_to_delete_np)
        
        # 清理docstore中的对应条目
        for doc_id in chunks_list:
            del store.docstore._dict[doc_id]
        
        # 更新index_to_docstore_id映射
        new_index_to_docstore = [doc_id for idx, doc_id in enumerate(store.index_to_docstore_id) if idx not in indices_to_delete]
        store.index_to_docstore_id = new_index_to_docstore
        
        # 重新保存到本地,确保修改持久化
        store.save_local(FAISS_USERGUIDE_INDEX)
        print(f"已成功删除文档{document}的所有chunk")
    else:
        print(f"未找到文档{document}对应的chunk")
完整代码示例
import pandas as pd
import numpy as np
from langchain.vectorstores import FAISS
from langchain.embeddings import YourEmbeddingModel  # 替换为你的Embedding模型

FAISS_USERGUIDE_INDEX = "your_faiss_index_path"  # 替换为你的FAISS索引路径
embeddings = YourEmbeddingModel()  # 初始化你的Embedding模型

db = FAISS.load_local(FAISS_USERGUIDE_INDEX, embeddings)

def store_to_df(store):
    v_dict = store.docstore._dict
    data_rows = []
    for k in v_dict.keys():
        doc_name = v_dict[k].metadata['source'].split('/')[-1]
        page_number = v_dict[k].metadata['page'] + 1
        content = v_dict[k].page_content
        data_rows.append({"chunk_id": k, "document": doc_name, "page": page_number, "content": content})
    vector_df = pd.DataFrame(data_rows)
    return vector_df

def delete_document(store, document):
    vector_df = store_to_df(store)
    chunks_list = vector_df.loc[vector_df['document'] == document]['chunk_id'].tolist()
    
    indices_to_delete = []
    for idx, doc_id in enumerate(store.index_to_docstore_id):
        if doc_id in chunks_list:
            indices_to_delete.append(idx)
    
    if indices_to_delete:
        indices_to_delete_np = np.array(indices_to_delete, dtype=np.int64)
        indices_to_delete_np = np.sort(indices_to_delete_np)[::-1]
        store.index.remove_ids(indices_to_delete_np)
        
        for doc_id in chunks_list:
            del store.docstore._dict[doc_id]
        
        new_index_to_docstore = [doc_id for idx, doc_id in enumerate(store.index_to_docstore_id) if idx not in indices_to_delete]
        store.index_to_docstore_id = new_index_to_docstore
        
        store.save_local(FAISS_USERGUIDE_INDEX)
        print(f"已成功删除文档{document}的所有chunk")
    else:
        print(f"未找到文档{document}对应的chunk")

delete_document(db, "doc2.pdf")

内容的提问来源于stack exchange,提问作者hzgoku

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.11 23:55:15