Weaviate中BM25及各类搜索函数无法返回预期结果求助
Weaviate检索结果异常问题
问题描述
已在Weaviate中创建集合,并通过LlamaIndex将若干文档导入Weaviate数据库。使用默认搜索时始终返回错误文档;测试BM25搜索时,即便复制目标文档中的完整短语,高分结果仍为其他文档。
服务器配置信息
- Weaviate服务器版本:1.24.10
- 部署方式:Docker
- LlamaIndex版本:0.10.42
文档准备
- 目标文档:某IEEE论文PDF(本地存储),另有20份文档一同导入用于检索测试。
Python配置信息
导入语句
# Weaviate相关 import weaviate from weaviate.classes.config import Configure, VectorDistances, Property, DataType from weaviate.util import generate_uuid5 from weaviate.query import MetadataQuery # LlamaIndex相关 from llama_index.core import SimpleDirectoryReader, VectorStoreIndex, Settings, StorageContext from llama_index.vector_stores.weaviate import WeaviateVectorStore from llama_index.core.node_parser import SentenceSplitter
基于Weaviate创建索引代码
# 创建Weaviate集合 def create_collection(client, collection_name): client.collections.create( collection_name, vectorizer_config=Configure.Vectorizer.text2vec_transformers(), vector_index_config=Configure.VectorIndex.hnsw(distance_metric=VectorDistances.COSINE), reranker_config=Configure.Reranker.transformers(), inverted_index_config=Configure.inverted_index( bm25_b=0.7, bm25_k1=1.25, index_null_state=True, index_property_length=True, index_timestamps=True ), ) # 使用LlamaIndex创建索引 def create_weaviate_index(client, index_name, doc_folder): create_collection(client, index_name) vector_store = WeaviateVectorStore(weaviate_client=client, index_name=index_name, text_key="content") storage_context = StorageContext.from_defaults(vector_store=vector_store) index = VectorStoreIndex.from_documents([], storage_context=storage_context) documents = SimpleDirectoryReader(input_dir=doc_folder) nodes = node_parser.get_nodes_from_documents(documents) index.insert_nodes(nodes) return index client = weaviate.connect_to_local() index_name = "LlamaIndex" doc_folder = "/path/to/doc_folder" create_weaviate_index(client, index_name, doc_folder)
文档查询测试
使用LlamaIndex查询
query_engine = index.as_query_engine() question = "EMA was created in 2001 to?" # 取自目标文档中的片段 response = query_engine.query(question) print(response) for node in response.source_nodes: print(node.metadata) # 未返回包含该片段的目标文档
使用Weaviate混合搜索(alpha=0)
collection = client.collections.get("LlamaIndex") question = "EMA was created in 2001 to?" # 取自目标文档中的片段 query_vector = embed_model.get_query_embedding(question) response = collection.query.hybrid( query=question, vector=query_vector, limit=5, alpha=0, return_metadata=MetadataQuery( distance=True, certainty=True, score=True, explain_score=True ) ) for obj in response.objects: print(f"METADATA: {obj.metadata}") # 未返回包含该片段的目标文档
使用Weaviate BM25搜索
collection = client.collections.get("LlamaIndex") question = "EMA was created in 2001 to?" # 取自目标文档中的片段 response = collection.query.bm25( query=question, limit=5, return_metadata=MetadataQuery( distance=True, certainty=True, score=True, explain_score=True ) ) for obj in response.objects: print(f"METADATA: {obj.metadata}") # 未返回包含该片段的目标文档
使用Weaviate近文本搜索
collection = client.collections.get("LlamaIndex") question = "EMA was created in 2001 to?" # 取自目标文档中的片段 response = collection.query.near_text( query=question, limit=5, return_metadata=MetadataQuery( distance=True, certainty=True, score=True, explain_score=True ) ) for obj in response.objects: print(f"METADATA: {obj.metadata}") # 未返回包含该片段的目标文档
内容的提问来源于stack exchange,提问作者SoftwearEnginear
相关产品推荐
相关产品推荐

