You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Scikit-learn与XGBoost分群模型训练中AttributeError问题排查求助

Scikit-learn与XGBoost分群模型训练中AttributeError问题排查求助

我现在遇到了一个在使用Scikit-learn和XGBoost进行分群模型训练时的AttributeError问题,折腾了好久都没解决,想请大家帮忙看看。

我的需求是针对6个不同的segment训练二分类模型,用GridSearchCV调参,同时保存模型和评估指标。但运行代码时一直报错,下面是我的完整代码、错误栈以及已经尝试过的解决方法:

训练与评估代码

# Import libraries
import pandas as pd
from sklearn.model_selection import train_test_split, GridSearchCV, KFold
from xgboost import XGBRegressor
from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score
import joblib

# Develop function to train and evaluate models
def train_and_evaluate_models(input_train_csv, models_dir, metrics_output_csv):
    # Load the training data
    train_data = pd.read_csv(input_train_csv)

    # Extract features and target
    X_train = train_data.drop(columns=['Response_ID', 'segment'])
    y_train = train_data['segment']

    metrics_list = []

    # Define hyperparameters to tune
    param_grid = {
        'max_depth': [3, 5, 7],
        'n_estimators': [100, 200, 300],
        'learning_rate': [0.01, 0.1, 0.2],
        'subsample': [0.8, 1.0],
        'colsample_bytree': [0.8, 1.0]
    }

    # Train, save models, and evaluate for each segment
    for segment in range(1, 7):
        # Binary target for the current segment
        y_train_segment = (y_train == segment).astype(int)

        # Split the data for evaluation
        X_train_split, X_valid, y_train_split, y_valid = train_test_split(X_train, y_train_segment, test_size=0.2, random_state=42)

        # Initialize the model
        model = XGBRegressor(
            objective='binary:logistic',
            eval_metric='logloss',
            use_label_encoder=False,
            early_stopping_rounds=10
        )

        # Hyperparameter tuning using GridSearchCV
        kf = KFold(n_splits=3, shuffle=True, random_state=42)
        grid_search = GridSearchCV(estimator=model, param_grid=param_grid, cv=kf, scoring='roc_auc', verbose=1, n_jobs=-1)

        # Fit the model with early stopping
        grid_search.fit(X_train_split, y_train_split, 
                        eval_set=[(X_valid, y_valid)], 
                        verbose=False)

        # Get the best model from grid search
        best_model = grid_search.best_estimator_

        # Save the best model to disk
        model_filename = f'{models_dir}/segment_{segment}_model.joblib'
        joblib.dump(best_model, model_filename)
        print(f'Model for segment {segment} saved to {model_filename}')

        # Predict on validation set
        y_valid_pred = best_model.predict(X_valid)
        y_valid_pred_binary = [1 if prob > 0.5 else 0 for prob in y_valid_pred]

        # Calculate evaluation metrics
        accuracy = accuracy_score(y_valid, y_valid_pred_binary)
        precision = precision_score(y_valid, y_valid_pred_binary)
        recall = recall_score(y_valid, y_valid_pred_binary)
        f1 = f1_score(y_valid, y_valid_pred_binary)
        auc = roc_auc_score(y_valid, y_valid_pred)

        metrics_list.append({
            'Segment': segment,
            'Accuracy': accuracy,
            'Precision': precision,
            'Recall': recall,
            'F1 Score': f1,
            'AUC': auc
        })

    # Save the metrics to a CSV file
    metrics_df = pd.DataFrame(metrics_list)
    metrics_df.to_csv(metrics_output_csv, index=False)

    return metrics_df

# Pathways
input_train_csv = r"C:\\Users\\me\\input.csv"
models_dir = r"C:\\Users\\me"
metrics_output_csv = r"C:\\Users\\me\\output.csv"

# Train, evaluate, and save models
metrics_df = train_and_evaluate_models(input_train_csv, models_dir, metrics_output_csv)
print(metrics_df)

报错信息

C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\sklearn\utils\_tags.py:354: FutureWarning: The XGBRegressor or classes from which it inherits use `_get_tags` and `_more_tags`. Please define the `__sklearn_tags__` method, or inherit from `sklearn.base.BaseEstimator` and/or other appropriate mixins such as `sklearn.base.TransformerMixin`, `sklearn.base.ClassifierMixin`, `sklearn.base.RegressorMixin`, and `sklearn.base.OutlierMixin`. From scikit-learn 1.7, not defining `__sklearn_tags__` will raise an error.
  warnings.warn(
---------------------------------------------------------------------------
AttributeError                            Traceback (most recent call last)
File C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\spyder_kernels\customize\utils.py:209, in exec_encapsulate_locals(code_ast, globals, locals, exec_fun, filename)
    207     if filename is None:
    208         filename = "<stdin>"
--> 209     exec_fun(compile(code_ast, filename, "exec"), globals, None)
    210 finally:
    211     if use_locals_hack:
    212         # Cleanup code

File c:\users\me\pyhton programs\2-model training\2_model_training.py:93
     90 metrics_output_csv = r"C:\\Users\\me\\model_metrics.csv"
     92 # Train, evaluate, and save models
--> 93 metrics_df = train_and_evaluate_models(input_train_csv, models_dir, metrics_output_csv)
     94 print(metrics_df)

File c:\users\me\pyhton programs\2-model training\2_model_training.py:49, in train_and_evaluate_models(input_train_csv, models_dir, metrics_output_csv)
     46 grid_search = GridSearchCV(estimator=model, param_grid=param_grid, cv=kf, scoring='roc_auc', verbose=1, n_jobs=-1)
     48 # Fit the model with early stopping
--> 49 grid_search.fit(X_train_split, y_train_split, 
     50                 eval_set=[(X_valid, y_valid)], 
     51                 verbose=False)
     53 # Get the best model from grid search
     54 best_model = grid_search.best_estimator_

File C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\sklearn\base.py:1389, in _fit_context.<locals>.decorator.<locals>.wrapper(estimator, *args, **kwargs)
   1382     estimator._validate_params()
   1384 with config_context(
   1385     skip_parameter_validation=(
   1386         prefer_skip_nested_validation or global_skip_validation
   1387     )
   1388 ):
--> 1389     return fit_method(estimator, *args, **kwargs)

File C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\sklearn\model_selection\_search.py:932, in BaseSearchCV.fit(self, X, y, **params)
    928 params = _check_method_params(X, params=params)
    930 routed_params = self._get_routed_params_for_fit(params)
--> 932 cv_orig = check_cv(self.cv, y, classifier=is_classifier(estimator))
    933 n_splits = cv_orig.get_n_splits(X, y, **routed_params.splitter.split)
    935 base_estimator = clone(self.estimator)

File C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\sklearn\base.py:1237, in is_classifier(estimator)
   1230     warnings.warn(
   1231         f"passing a class to {print(inspect.stack()[0][3])} is deprecated and "
   1232         "will be removed in 1.8. Use an instance of the class instead.",
   1233         FutureWarning,
   1234     )
   1235     return getattr(estimator, "_estimator_type", None) == "classifier"
--> 1237 return get_tags(estimator).estimator_type == "classifier"

File C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\sklearn\utils\_tags.py:405, in get_tags(estimator)
    403 for klass in reversed(type(estimator).mro()):
    404     if "__sklearn_tags__" in vars(klass):
--> 405         sklearn_tags_provider[klass] = klass.__sklearn_tags__(estimator)  # type: ignore[attr-defined]
    406         class_order.append(klass)
    407     elif "_more_tags" in vars(klass):

File C:\ProgramData\spyder-6\envs\spyder-runtime\Lib\site-packages\sklearn\base.py:613, in RegressorMixin.__sklearn_tags__(self)
    612 def __sklearn_tags__(self):
--> 613     tags = super().__sklearn_tags__()
    614     tags.estimator_type = "regressor"
    615     tags.regressor_tags = RegressorTags()

AttributeError: 'super' object has no attribute '__sklearn_tags__'

已尝试的解决方法

  • 升级XGBoost到最新版本,也尝试过降级Scikit-learn版本,试图解决两者的兼容性问题,但错误依旧
  • 检查输入数据的类型,确保所有特征列都是数值型(int/float),没有字符串类型的特征,排除数据类型导致的问题
  • 尝试过多种针对Scikit-learn和XGBoost兼容性的workaround,但都没有解决这个AttributeError

麻烦各位帮忙看看问题出在哪里,应该怎么解决?

备注:内容来源于stack exchange,提问作者Sdeb

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.14 18:13:09