如何修复Python中Ragas库的AttributeError:'str'无'generator_llm'属性
解决Ragas脚本运行时的AttributeError问题
问题背景
同一份Ragas生成测试集的代码在Notebook中可正常运行,但转为脚本或在Notebook中导入脚本调用时,执行到100%后报错,报错信息显示'str' object has no attribute 'generator_llm'。脚本与Notebook使用的Ragas版本均为0.1.4。
原代码
import importlib.resources import os import pandas as pd import yaml from dotenv import load_dotenv from google.cloud import storage from google.cloud.storage import Blob from langchain.docstore.document import Document from langchain_community.document_loaders import PDFPlumberLoader from langchain_experimental.text_splitter import SemanticChunker from langchain_openai.embeddings import OpenAIEmbeddings from ragas.testset.evolutions import multi_context, reasoning, simple from ragas.testset.generator import TestsetGenerator from tqdm.notebook import tqdm load_dotenv() def get_pdf_files(collection_name, filegroup_name): """Get the pdf files from the collection and filegroup from the ds-librairie-genai-provisioning""" collections_path = importlib.resources.files("genai_provisioning").joinpath( "collections" ) collection_path = collections_path.joinpath(collection_name) filegroups_path = collection_path.joinpath("filegroups") files_directory = filegroups_path.joinpath(filegroup_name).joinpath("files") pdf_files = list(files_directory.glob("*.pdf")) return pdf_files def combine_all_page(docs): """Combine all pages from one PDF into one Document object""" try: page_content = "/n".join([doc.page_content for doc in docs]) page_metadata = docs[0].metadata if "page" in page_metadata: page_metadata.pop("page") return [Document(page_content=page_content, metadata=page_metadata)] except Exception as e: print(e) def generate_testset(configs_file): """Generate RAGAS synthetic test set""" with open(configs_file) as f: configs = yaml.safe_load(f) collection_name = configs["collection_name"] filegroup_name = configs["filegroup_name"] version = configs["version"] files = get_pdf_files(collection_name, filegroup_name) print(f"Number of files: {len(files)}") docs = [] for file in tqdm(files): doc_ = PDFPlumberLoader( str(file), text_kwargs={ "x_tolerance": configs["x_tolerance"], "y_tolerance": configs["y_tolerance"], }, ).load() doc_ = combine_all_page(doc_) docs.extend(doc_) print(f"Number of documents: {len(docs)}") for doc_ in docs: doc_.metadata["filename"] = doc_.metadata["source"] splitted_testset_docs = docs # RAGAS to generate a synthetic testset generator = TestsetGenerator.with_openai() testset = generator.generate_with_langchain_docs( splitted_testset_docs, test_size=configs["test_size"], with_debugging_logs=True, distributions=configs["distributions"], )
报错信息
--------------------------------------------------------------------------- AttributeError Traceback (most recent call last) Cell In[2], line 3 1 from generate_testset import generate_testset ----> 3 generate_testset("generate-testset.yml") File ~/github_repos/ds-research-genai/common/src/rag_pipeline_evaluator/generate_testset.py:86, in generate_testset(configs_file) 83 print("len(splitted_testset_docs)", len(splitted_testset_docs)) 84 print("splitted_testset_docs[0]", splitted_testset_docs[0]) ---> 86 testset = generator.generate_with_langchain_docs( 87 splitted_testset_docs, 88 test_size=configs["test_size"], 89 with_debugging_logs=True, 90 distributions=configs["distributions"], 91 ) 93 # Remove nan and short ground truth 94 testset_improved = testset File ~/github_repos/ds-research-genai/common/src/rag_pipeline_evaluator/venv/lib/python3.11/site-packages/ragas/testset/generator.py:179, in TestsetGenerator.generate_with_langchain_docs(self, documents, test_size, distributions, with_debugging_logs, is_async, raise_exceptions, run_config) 174 # chunk documents and add to docstore 175 self.docstore.add_documents( 176 [Document.from_langchain_document(doc) for doc in documents] 177 ) --> 179 return self.generate( 180 test_size=test_size, 181 distributions=distributions, 182 with_debugging_logs=with_debugging_logs, 183 is_async=is_async, 184 raise_exceptions=raise_exceptions, 185 run_config=run_config, 186 ) File ~/github_repos/ds-research-genai/common/src/rag_pipeline_evaluator/venv/lib/python3.11/site-packages/ragas/testset/generator.py:227, in TestsetGenerator.generate(self, test_size, distributions, with_debugging_logs, is_async, raise_exceptions, run_config) 225 # init filters and evolutions 226 for evolution in distributions: --> 227 self.init_evolution(evolution) 228 evolution.init(is_async=is_async, run_config=run_config) 230 if with_debugging_logs: File ~/github_repos/ds-research-genai/common/src/rag_pipeline_evaluator/venv/lib/python3.11/site-packages/ragas/testset/generator.py:189, in TestsetGenerator.init_evolution(self, evolution) 188 def init_evolution(self, evolution: Evolution) -> None: --> 189 if evolution.generator_llm is None: 190 evolution.generator_llm = self.generator_llm 191 if evolution.docstore is None: AttributeError: 'str' object has no attribute 'generator_llm'
问题原因
- distributions类型不匹配:YAML配置文件中
distributions存储的是字符串(如["simple", "reasoning"]),但Ragas的generate_with_langchain_docs方法需要传入的是从ragas.testset.evolutions导入的Evolution类实例。Notebook中可能直接使用了对象而非从YAML读取字符串,因此未触发该问题。 - tqdm导入错误:脚本中使用
from tqdm.notebook import tqdm,该模块仅适用于Notebook环境,脚本运行时会引发界面相关错误。
修复方案
1. 转换distributions字符串为对应Evolution对象
在加载YAML配置后,添加字符串到Evolution对象的映射逻辑:
# 在generate_testset函数中,加载configs后添加 evolution_map = { "simple": simple, "reasoning": reasoning, "multi_context": multi_context } # 将配置中的字符串转为对应的Evolution实例 configs["distributions"] = [evolution_map[evo] for evo in configs["distributions"]]
2. 替换tqdm导入方式
将脚本中的tqdm.notebook替换为普通tqdm:
# 原导入 # from tqdm.notebook import tqdm # 替换为 from tqdm import tqdm
修复后的完整generate_testset函数
def generate_testset(configs_file): """Generate RAGAS synthetic test set""" with open(configs_file) as f: configs = yaml.safe_load(f) # 新增:转换distributions字符串为Evolution对象 evolution_map = { "simple": simple, "reasoning": reasoning, "multi_context": multi_context } configs["distributions"] = [evolution_map[evo] for evo in configs["distributions"]] collection_name = configs["collection_name"] filegroup_name = configs["filegroup_name"] version = configs["version"] files = get_pdf_files(collection_name, filegroup_name) print(f"Number of files: {len(files)}") docs = [] for file in tqdm(files): doc_ = PDFPlumberLoader( str(file), text_kwargs={ "x_tolerance": configs["x_tolerance"], "y_tolerance": configs["y_tolerance"], }, ).load() doc_ = combine_all_page(doc_) docs.extend(doc_) print(f"Number of documents: {len(docs)}") for doc_ in docs: doc_.metadata["filename"] = doc_.metadata["source"] splitted_testset_docs = docs # RAGAS to generate a synthetic testset generator = TestsetGenerator.with_openai() testset = generator.generate_with_langchain_docs( splitted_testset_docs, test_size=configs["test_size"], with_debugging_logs=True, distributions=configs["distributions"], )
验证
修改后,无论是终端运行脚本还是在Notebook中导入脚本调用,代码均可正常生成测试集,不再触发AttributeError。
内容的提问来源于stack exchange,提问作者Nicolas Ferland
相关产品推荐
相关产品推荐

