KNN分类模型报错:'NoneType' object has no attribute 'split'
问题:KNeighborsClassifier调用predict时触发AttributeError
此前使用Logistic Regression和Decision Tree Classifier均正常运行,但使用KNN分类模型时,调用predict方法出现AttributeError。
代码片段
from sklearn.neighbors import KNeighborsClassifier KNN = KNeighborsClassifier().fit(X_train,Y_train) train_preds5 = KNN.predict(X_train) print("Model accuracy on train is: ", accuracy_score(Y_train, train_preds5))
报错信息
AttributeError Traceback (most recent call last) Cell In[87], line 5 3 KNN = KNeighborsClassifier().fit(X_train,Y_train) 4 #predict on train ----> 5 train_preds5 = KNN.predict(X_train) 6 #accuracy on train 7 print("Model accuracy on train is: ", accuracy_score(Y_train, train_preds5)) File ~\anaconda3\Lib\site-packages\sklearn\neighbors\_classification.py:234, in KNeighborsClassifier.predict(self, X) 218 """Predict the class labels for the provided data. 219 220 Parameters (...) 229 Class labels for each data sample. 230 """ 231 if self.weights == "uniform": 232 # In that case, we do not need the distances to perform 233 # the weighting so we do not compute them. ---> 234 neigh_ind = self.kneighbors(X, return_distance=False) 235 neigh_dist = None 236 else: File ~\anaconda3\Lib\site-packages\sklearn\neighbors\_base.py:824, in KNeighborsMixin.kneighbors(self, X, n_neighbors, return_distance) 817 use_pairwise_distances_reductions = ( 818 self._fit_method == "brute" 819 and ArgKmin.is_usable_for( 820 X if X is not None else self._fit_X, self._fit_X, self.effective_metric_ 821 ) 822 ) 823 if use_pairwise_distances_reductions: ---> 824 results = ArgKmin.compute( 825 X=X, 826 Y=self._fit_X, 827 k=n_neighbors, 828 metric=self.effective_metric_, 829 metric_kwargs=self.effective_metric_params_, 830 strategy="auto", 831 return_distance=return_distance, 832 ) 834 elif ( 835 self._fit_method == "brute" and self.metric == "precomputed" and issparse(X) 836 ): 837 results = _kneighbors_from_graph( 838 X, n_neighbors=n_neighbors, return_distance=return_distance 839 ) File ~\anaconda3\Lib\site-packages\sklearn\metrics\_pairwise_distances_reduction\_dispatcher.py:277, in ArgKmin.compute(cls, X, Y, k, metric, chunk_size, metric_kwargs, strategy, return_distance) 196 """Compute the argkmin reduction. 197 198 Parameters (...) 274 returns. 275 """ 276 if X.dtype == Y.dtype == np.float64: ---> 277 return ArgKmin64.compute( 278 X=X, 279 Y=Y, 280 k=k, 281 metric=metric, 282 chunk_size=chunk_size, 283 metric_kwargs=metric_kwargs, 284 strategy=strategy, 285 return_distance=return_distance, 286 ) 288 if X.dtype == Y.dtype == np.float32: 289 return ArgKmin32.compute( 290 X=X, 291 Y=Y, (...) 297 return_distance=return_distance, 298 ) File sklearn\metrics\_pairwise_distances_reduction\_argkmin.pyx:95, in sklearn.metrics._pairwise_distances_reduction._argkmin.ArgKmin64.compute() File ~\anaconda3\Lib\site-packages\sklearn\utils\fixes.py:139, in threadpool_limits(limits, user_api) 137 return controller.limit(limits=limits, user_api=user_api) 138 else: ---> 139 return threadpoolctl.threadpool_limits(limits=limits, user_api=user_api) File ~\anaconda3\Lib\site-packages\threadpoolctl.py:171, in threadpool_limits.__init__(self, limits, user_api) 167 def __init__(self, limits=None, user_api=None): 168 self._limits, self._user_api, self._prefixes = \ 169 self._check_params(limits, user_api) ---> 171 self._original_info = self._set_threadpool_limits() File ~\anaconda3\Lib\site-packages\threadpoolctl.py:268, in threadpool_limits._set_threadpool_limits(self) 265 if self._limits is None: 266 return None ---> 268 modules = _ThreadpoolInfo(prefixes=self._prefixes, 269 user_api=self._user_api) 270 for module in modules: 271 # self._limits is a dict {key: num_threads} where key is either 272 # a prefix or a user_api. If a module matches both, the limit 273 # corresponding to the prefix is chosed. 274 if module.prefix in self._limits: File ~\anaconda3\Lib\site-packages\threadpoolctl.py:340, in _ThreadpoolInfo.__init__(self, user_api, prefixes, modules) 337 self.user_api = [] if user_api is None else user_api 339 self.modules = [] ---> 340 self._load_modules() 341 self._warn_if_incompatible_openmp() 342 else: File ~\anaconda3\Lib\site-packages\threadpoolctl.py:373, in _ThreadpoolInfo._load_modules(self) 371 self._find_modules_with_dyld() 372 elif sys.platform == "win32": ---> 373 self._find_modules_with_enum_process_module_ex() 374 else: 375 self._find_modules_with_dl_iterate_phdr() File ~\anaconda3\Lib\site-packages\threadpoolctl.py:485, in _ThreadpoolInfo._find_modules_with_enum_process_module_ex(self) 482 filepath = buf.value 484 # Store the module if it is supported and selected ---> 485 self._make_module_from_path(filepath) 486 finally: 487 kernel_32.CloseHandle(h_process) File ~\anaconda3\Lib\site-packages\threadpoolctl.py:515, in _ThreadpoolInfo._make_module_from_path(self, filepath) 513 if prefix in self.prefixes or user_api in self.user_api: 514 module_class = globals()[module_class] ---> 515 module = module_class(filepath, prefix, user_api, internal_api) 516 self.modules.append(module) File ~\anaconda3\Lib\site-packages\threadpoolctl.py:606, in _Module.__init__(self, filepath, prefix, user_api, internal_api) 604 self.internal_api = internal_api 605 self._dynlib = ctypes.CDLL(filepath, mode=_RTLD_NOLOAD) ---> 606 self.version = self.get_version() 607 self.num_threads = self.get_num_threads() 608 self._get_extra_info() File ~\anaconda3\Lib\site-packages\threadpoolctl.py:646, in _OpenBLASModule.get_version(self) 643 get_config = getattr(self._dynlib, "openblas_get_config", 644 lambda: None) 645 get_config.restype = ctypes.c_char_p ---> 646 config = get_config().split() 647 if config[0] == b"OpenBLAS": 648 return config[1].decode("utf-8") AttributeError: 'NoneType' object has no attribute 'split'
期望输出训练集准确率,但触发上述属性错误。
解决方案
这个错误本质是threadpoolctl库在读取OpenBLAS版本时出现问题,和KNN模型本身无关,以下是几种可行的修复方法:
临时禁用线程池限制:在代码开头添加以下代码,跳过threadpoolctl的线程限制逻辑:
import os os.environ["OMP_NUM_THREADS"] = "1" os.environ["OPENBLAS_NUM_THREADS"] = "1"降级/升级threadpoolctl库:
卸载当前版本,安装指定版本:pip uninstall threadpoolctl -y pip install threadpoolctl==3.1.03.1.0版本修复了部分Windows环境下OpenBLAS版本读取的bug。
指定KNN的计算方法为brute并禁用距离优化:
初始化KNN时显式设置algorithm='brute',并关闭pairwise距离优化:from sklearn.neighbors import KNeighborsClassifier from sklearn.utils._openmp_helpers import _openmp_effective_n_threads # 临时设置线程数 _openmp_effective_n_threads.cache_clear() KNN = KNeighborsClassifier(algorithm='brute').fit(X_train,Y_train) train_preds5 = KNN.predict(X_train) print("Model accuracy on train is: ", accuracy_score(Y_train, train_preds5))更新Anaconda环境:
如果使用Anaconda,更新整个环境的依赖包,修复OpenBLAS和threadpoolctl的兼容性问题:conda update --all
内容的提问来源于stack exchange,提问作者Karthik Kumar P
相关产品推荐
相关产品推荐

