升级Python至3.10.8后无法加载HuggingFace Tokenizer的解决方法
解决Python 3.10下加载HuggingFace Tokenizer时的
collections.MutableMapping错误 问题背景
Python版本升级至3.10.8(JupyterLab环境),重装依赖后,使用transformers 4.24.0加载deepset/xlm-roberta-large-squad2的tokenizer时触发错误。
执行代码:
# Import libraries from transformers import pipeline, AutoTokenizer # Define checkpoint model_checkpoint = 'deepset/xlm-roberta-large-squad2' # Tokenizer tokenizer = AutoTokenizer.from_pretrained(model_checkpoint)
报错信息:
--------------------------------------------------------------------------- AttributeError Traceback (most recent call last) Cell In [3], line 2 1 # Tokenizer ----> 2 tokenizer = AutoTokenizer.from_pretrained(model_checkpoint) File ~/.local/lib/python3.10/site-packages/transformers/models/auto/tokenization_auto.py:637, in AutoTokenizer.from_pretrained(cls, pretrained_model_name_or_path, *inputs, **kwargs) 635 tokenizer_class_py, tokenizer_class_fast = TOKENIZER_MAPPING[type(config)] 636 if tokenizer_class_fast and (use_fast or tokenizer_class_py is None): ---> 637 return tokenizer_class_fast.from_pretrained(pretrained_model_name_or_path, *inputs, **kwargs) 638 else: 639 if tokenizer_class_py is not None: File ~/.local/lib/python3.10/site-packages/transformers/tokenization_utils_base.py:1777, in PreTrainedTokenizerBase.from_pretrained(cls, pretrained_model_name_or_path, *init_inputs, **kwargs) 1774 else: 1775 logger.info(f"loading file {file_path} from cache at {resolved_vocab_files[file_id]}") -> 1777 return cls._from_pretrained( 1778 resolved_vocab_files, 1779 pretrained_model_name_or_path, 1780 init_configuration, 1781 *init_inputs, 1782 use_auth_token=use_auth_token, 1783 cache_dir=cache_dir, 1784 local_files_only=local_files_only, 1785 _commit_hash=commit_hash, 1786 **kwargs, 1787 ) File ~/.local/lib/python3.10/site-packages/transformers/tokenization_utils_base.py:1932, in PreTrainedTokenizerBase._from_pretrained(cls, resolved_vocab_files, pretrained_model_name_or_path, init_configuration, use_auth_token, cache_dir, local_files_only, _commit_hash, *init_inputs, **kwargs) 1930 # Instantiate tokenizer. 1931 try: -> 1932 tokenizer = cls(*init_inputs, **init_kwargs) 1933 except OSError: 1934 raise OSError( 1935 "Unable to load vocabulary from file. " 1936 "Please check that the provided vocabulary is accessible and not corrupted." 1937 ) File ~/.local/lib/python3.10/site-packages/transformers/models/xlm_roberta/tokenization_xlm_roberta_fast.py:155, in XLMRobertaTokenizerFast.__init__(self, vocab_file, tokenizer_file, bos_token, eos_token, sep_token, cls_token, unk_token, pad_token, mask_token, **kwargs) 139 def __init__( 140 self, 141 vocab_file=None, (...) 151 ): 152 # Mask token behave like a normal word, i.e. include the space before it 153 mask_token = AddedToken(mask_token, lstrip=True, rstrip=False) if isinstance(mask_token, str) else mask_token ---> 155 super().__init__( 156 vocab_file, 157 tokenizer_file=tokenizer_file, 158 bos_token=bos_token, 159 eos_token=eos_token, 160 sep_token=sep_token, 161 cls_token=cls_token, 162 unk_token=unk_token, 163 pad_token=pad_token, 164 mask_token=mask_token, 165 **kwargs, 166 ) 168 self.vocab_file = vocab_file 169 self.can_save_slow_tokenizer = False if not self.vocab_file else True File ~/.local/lib/python3.10/site-packages/transformers/tokenization_utils_fast.py:114, in PreTrainedTokenizerFast.__init__(self, *args, **kwargs) 111 fast_tokenizer = TokenizerFast.from_file(fast_tokenizer_file) 112 elif slow_tokenizer is not None: 113 # We need to convert a slow tokenizer to build the backend ---> 114 fast_tokenizer = convert_slow_tokenizer(slow_tokenizer) 115 elif self.slow_tokenizer_class is not None: 116 # We need to create and convert a slow tokenizer to build the backend 117 slow_tokenizer = self.slow_tokenizer_class(*args, **kwargs) File ~/.local/lib/python3.10/site-packages/transformers/convert_slow_tokenizer.py:1162, in convert_slow_tokenizer(transformer_tokenizer) 1154 raise ValueError( 1155 f"An instance of tokenizer class {tokenizer_class_name} cannot be converted in a Fast tokenizer instance." 1156 " No converter was found. Currently available slow->fast convertors:" 1157 f" {list(SLOW_TO_FAST_CONVERTERS.keys())}" 1158 ) 1160 converter_class = SLOW_TO_FAST_CONVERTERS[tokenizer_class_name] -> 1162 return converter_class(transformer_tokenizer).converted() File ~/.local/lib/python3.10/site-packages/transformers/convert_slow_tokenizer.py:438, in SpmConverter.__init__(self, *args) 434 requires_backends(self, "protobuf") 436 super().__init__(*args) ---> 438 from .utils import sentencepiece_model_pb2 as model_pb2 440 m = model_pb2.ModelProto() 441 with open(self.original_tokenizer.vocab_file, "rb") as f: File ~/.local/lib/python3.10/site-packages/transformers/utils/sentencepiece_model_pb2.py:20 18 from google.protobuf import descriptor as _descriptor 19 from google.protobuf import message as _message ---> 20 from google.protobuf import reflection as _reflection 21 from google.protobuf import symbol_database as _symbol_database 24 # @@protoc_insertion_point(imports) File /usr/lib/python3/dist-packages/google/protobuf/reflection.py:58 56 from google.protobuf.pyext import cpp_message as message_impl 57 else: ---> 58 from google.protobuf.internal import python_message as message_impl 60 # The type of all Message classes. 61 # Part of the public interface, but normally only used by message factories. 62 GeneratedProtocolMessageType = message_impl.GeneratedProtocolMessageType File /usr/lib/python3/dist-packages/google/protobuf/internal/python_message.py:69 66 import copyreg as copyreg 68 # We use "as" to avoid name collisions with variables. ---> 69 from google.protobuf.internal import containers 70 from google.protobuf.internal import decoder 71 from google.protobuf.internal import encoder File /usr/lib/python3/dist-packages/google/protobuf/internal/containers.py:182 177 collections.MutableMapping.register(MutableMapping) 179 else: 180 # In Python 3 we can just use MutableMapping directly, because it defines 181 # __slots__. ---> 182 MutableMapping = collections.MutableMapping 185 class BaseContainer(object): 187 """Base container class.""" AttributeError: module 'collections' has no attribute 'MutableMapping'
错误原因
Python 3.10起,collections模块中的抽象基类(如MutableMapping)被迁移到collections.abc子模块,而当前系统安装的google.protobuf版本过旧,仍在使用旧的导入路径,导致触发AttributeError。
解决方案
1. 升级google.protobuf(推荐)
执行以下命令升级到兼容Python 3.10的版本:
pip install --upgrade google.protobuf
新版本的protobuf已适配Python 3.10的collections.abc路径,升级后即可解决问题。
2. 修改旧版protobuf源码(临时修复)
若因依赖限制无法升级,找到报错文件/usr/lib/python3/dist-packages/google/protobuf/internal/containers.py,修改第182行:
将
MutableMapping = collections.MutableMapping
替换为:
try: from collections.abc import MutableMapping except ImportError: from collections import MutableMapping
该修改可同时兼容新旧Python版本。
3. 绕过fast tokenizer转换
加载tokenizer时添加use_fast=False参数,避免触发fast tokenizer的转换流程(无需用到protobuf相关代码):
tokenizer = AutoTokenizer.from_pretrained(model_checkpoint, use_fast=False)
此方法可快速绕开错误,但tokenizer运行速度可能会有所下降。
内容的提问来源于stack exchange,提问作者SilentCloud
相关产品推荐
相关产品推荐

