构建通用MLP架构:实现适配任意输入数据集的动态模型
Alright, let's figure out how to turn your existing dataset-specific model into a dynamic, universal pipeline that works with any arbitrary input dataset. You mentioned you have an MLP model already—we'll make sure this system can integrate that too, while reusing your existing logic where possible. Here's a step-by-step solution:
The core goal is to refactor your hardcoded Iris pipeline into modular, parameterized components that adapt to any dataset structure, task type (classification/regression), and model choice.
Step 1: Refactored Universal Code
This version handles automatic preprocessing, task-agnostic model evaluation, and full pipeline persistence (so your saved model can process raw data directly later):
import pandas as pd import matplotlib.pyplot as plt from sklearn import model_selection from sklearn.metrics import ( classification_report, confusion_matrix, accuracy_score, mean_squared_error, r2_score ) from sklearn.preprocessing import StandardScaler, OneHotEncoder from sklearn.compose import ColumnTransformer from sklearn.pipeline import Pipeline from sklearn.linear_model import LogisticRegression from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor from sklearn.neighbors import KNeighborsClassifier, KNeighborsRegressor from sklearn.discriminant_analysis import LinearDiscriminantAnalysis from sklearn.naive_bayes import GaussianNB from sklearn.svm import SVC, SVR from sklearn.ensemble import RandomForestRegressor from sklearn.neural_network import MLPClassifier, MLPRegressor # Add your MLP here from joblib import dump, load def load_and_preprocess_data(data_path, target_col, feature_cols=None): """ Load any CSV dataset and handle basic preprocessing for universal compatibility. Automatically handles numerical/categorical features and missing values. """ df = pd.read_csv(data_path) # Use all columns except target if no features are specified if not feature_cols: feature_cols = [col for col in df.columns if col != target_col] X = df[feature_cols] y = df[target_col] # Split features into numerical and categorical types num_features = X.select_dtypes(include=['int64', 'float64']).columns cat_features = X.select_dtypes(include=['object', 'category']).columns # Build preprocessing pipeline: scale numbers, encode categories preprocessor = ColumnTransformer( transformers=[ ('num', StandardScaler(), num_features), ('cat', OneHotEncoder(handle_unknown='ignore'), cat_features) ]) return X, y, preprocessor def split_dataset(X, y, test_size=0.2, seed=7): """Split data into training and validation sets with consistent randomness.""" return model_selection.train_test_split(X, y, test_size=test_size, random_state=seed) def get_model_list(task_type='classification'): """ Return a task-specific list of models (add your existing MLP here!). Extend this with any custom models you want to test. """ models = [] if task_type == 'classification': models.append(('LR', LogisticRegression())) models.append(('LDA', LinearDiscriminantAnalysis())) models.append(('KNN', KNeighborsClassifier())) models.append(('CART', DecisionTreeClassifier())) models.append(('NB', GaussianNB())) models.append(('SVM', SVC())) models.append(('MLP', MLPClassifier(hidden_layer_sizes=(100,), max_iter=500))) # Your MLP elif task_type == 'regression': models.append(('KNN', KNeighborsRegressor())) models.append(('CART', DecisionTreeRegressor())) models.append(('SVM', SVR())) models.append(('RF', RandomForestRegressor())) models.append(('MLP', MLPRegressor(hidden_layer_sizes=(100,), max_iter=500))) # Regression MLP return models def evaluate_models(X_train, y_train, models, preprocessor, scoring, seed=7): """Evaluate multiple models with cross-validation and print/plot results.""" results = [] names = [] for name, model in models: # Wrap model with preprocessing to ensure fair comparison pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', model)]) kfold = model_selection.KFold(n_splits=10, random_state=seed, shuffle=True) cv_results = model_selection.cross_val_score(pipeline, X_train, y_train, cv=kfold, scoring=scoring) results.append(cv_results) names.append(name) msg = f"{name}: Mean Score = {cv_results.mean():.4f} | Std Dev = {cv_results.std():.4f}" print(msg) # Plot model performance comparison fig = plt.figure(figsize=(10,6)) fig.suptitle(f'Model Performance Comparison ({scoring})') ax = fig.add_subplot(111) plt.boxplot(results) ax.set_xticklabels(names) plt.show() return results, names def train_and_save_best_model(X_train, y_train, X_val, y_val, models, preprocessor, task_type, save_path='best_universal_model.pkl'): """Train the top-performing model and save the full preprocessing+model pipeline.""" # Get cross-validation results to find the best model results, names = evaluate_models(X_train, y_train, models, preprocessor, scoring='accuracy' if task_type == 'classification' else 'r2') best_model_idx = results.index(max(results)) best_model_name = names[best_model_idx] best_model = next(model for name, model in models if name == best_model_name) # Build full pipeline (preprocessing + model) full_pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', best_model)]) full_pipeline.fit(X_train, y_train) # Evaluate on validation set predictions = full_pipeline.predict(X_val) print("\n--- Best Model Validation Results ---") if task_type == 'classification': print(f"Accuracy: {accuracy_score(y_val, predictions):.4f}") print("\nClassification Report:\n", classification_report(y_val, predictions)) print("Confusion Matrix:\n", confusion_matrix(y_val, predictions)) elif task_type == 'regression': print(f"RMSE: {mean_squared_error(y_val, predictions, squared=False):.4f}") print(f"R2 Score: {r2_score(y_val, predictions):.4f}") # Save the entire pipeline (critical for handling raw data later) dump(full_pipeline, save_path) print(f"\nModel pipeline saved to {save_path}") return full_pipeline # ------------------------------ # Example: Test with Iris Dataset # ------------------------------ if __name__ == "__main__": # Configure these variables for YOUR dataset! DATA_PATH = "iris.data" TARGET_COL = 'class' FEATURE_COLS = ['sepal-length', 'sepal-width', 'petal-length', 'petal-width'] TASK_TYPE = 'classification' SAVE_PATH = 'iris_universal_model.pkl' # Run the full pipeline X, y, preprocessor = load_and_preprocess_data(DATA_PATH, TARGET_COL, FEATURE_COLS) X_train, X_val, y_train, y_val = split_dataset(X, y) models = get_model_list(TASK_TYPE) train_and_save_best_model(X_train, y_train, X_val, y_val, models, preprocessor, TASK_TYPE, SAVE_PATH)
Step 2: Key Improvements for Universality
- Modular, Parameterized Functions: Every step (data loading, splitting, evaluation) accepts parameters, so you can swap datasets or tasks in seconds.
- Automatic Preprocessing: Handles numerical scaling and categorical encoding out of the box—no more manual data cleaning for new datasets.
- Task-Agnostic: Supports both classification and regression; just set
TASK_TYPEand the model list/scoring adjusts automatically. - Full Pipeline Saving: Saves preprocessing logic alongside the model, so when you load it later, it can process raw, unmodified data directly.
- Extensible: Add your existing MLP (or any custom model) to the
get_model_listfunction with one line of code.
Step 3: Use the Saved Model on New Data
To make predictions with your saved universal model:
loaded_pipeline = load('best_universal_model.pkl') # Example: Predict on raw, unprocessed data new_sample = pd.DataFrame({ 'sepal-length': [5.1], 'sepal-width': [3.5], 'petal-length': [1.4], 'petal-width': [0.2] }) prediction = loaded_pipeline.predict(new_sample) print(f"Predicted class: {prediction[0]}")
内容的提问来源于stack exchange,提问作者pippo pluto

