You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

构建通用MLP架构:实现适配任意输入数据集的动态模型

Alright, let's figure out how to turn your existing dataset-specific model into a dynamic, universal pipeline that works with any arbitrary input dataset. You mentioned you have an MLP model already—we'll make sure this system can integrate that too, while reusing your existing logic where possible. Here's a step-by-step solution:

Universal Dynamic Model Pipeline for Arbitrary Datasets

The core goal is to refactor your hardcoded Iris pipeline into modular, parameterized components that adapt to any dataset structure, task type (classification/regression), and model choice.

Step 1: Refactored Universal Code

This version handles automatic preprocessing, task-agnostic model evaluation, and full pipeline persistence (so your saved model can process raw data directly later):

import pandas as pd
import matplotlib.pyplot as plt
from sklearn import model_selection
from sklearn.metrics import (
    classification_report, confusion_matrix, accuracy_score,
    mean_squared_error, r2_score
)
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.linear_model import LogisticRegression
from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor
from sklearn.neighbors import KNeighborsClassifier, KNeighborsRegressor
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
from sklearn.naive_bayes import GaussianNB
from sklearn.svm import SVC, SVR
from sklearn.ensemble import RandomForestRegressor
from sklearn.neural_network import MLPClassifier, MLPRegressor  # Add your MLP here
from joblib import dump, load

def load_and_preprocess_data(data_path, target_col, feature_cols=None):
    """
    Load any CSV dataset and handle basic preprocessing for universal compatibility.
    Automatically handles numerical/categorical features and missing values.
    """
    df = pd.read_csv(data_path)
    
    # Use all columns except target if no features are specified
    if not feature_cols:
        feature_cols = [col for col in df.columns if col != target_col]
    
    X = df[feature_cols]
    y = df[target_col]
    
    # Split features into numerical and categorical types
    num_features = X.select_dtypes(include=['int64', 'float64']).columns
    cat_features = X.select_dtypes(include=['object', 'category']).columns
    
    # Build preprocessing pipeline: scale numbers, encode categories
    preprocessor = ColumnTransformer(
        transformers=[
            ('num', StandardScaler(), num_features),
            ('cat', OneHotEncoder(handle_unknown='ignore'), cat_features)
        ])
    
    return X, y, preprocessor

def split_dataset(X, y, test_size=0.2, seed=7):
    """Split data into training and validation sets with consistent randomness."""
    return model_selection.train_test_split(X, y, test_size=test_size, random_state=seed)

def get_model_list(task_type='classification'):
    """
    Return a task-specific list of models (add your existing MLP here!).
    Extend this with any custom models you want to test.
    """
    models = []
    if task_type == 'classification':
        models.append(('LR', LogisticRegression()))
        models.append(('LDA', LinearDiscriminantAnalysis()))
        models.append(('KNN', KNeighborsClassifier()))
        models.append(('CART', DecisionTreeClassifier()))
        models.append(('NB', GaussianNB()))
        models.append(('SVM', SVC()))
        models.append(('MLP', MLPClassifier(hidden_layer_sizes=(100,), max_iter=500)))  # Your MLP
    elif task_type == 'regression':
        models.append(('KNN', KNeighborsRegressor()))
        models.append(('CART', DecisionTreeRegressor()))
        models.append(('SVM', SVR()))
        models.append(('RF', RandomForestRegressor()))
        models.append(('MLP', MLPRegressor(hidden_layer_sizes=(100,), max_iter=500)))  # Regression MLP
    return models

def evaluate_models(X_train, y_train, models, preprocessor, scoring, seed=7):
    """Evaluate multiple models with cross-validation and print/plot results."""
    results = []
    names = []
    for name, model in models:
        # Wrap model with preprocessing to ensure fair comparison
        pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', model)])
        kfold = model_selection.KFold(n_splits=10, random_state=seed, shuffle=True)
        cv_results = model_selection.cross_val_score(pipeline, X_train, y_train, cv=kfold, scoring=scoring)
        results.append(cv_results)
        names.append(name)
        msg = f"{name}: Mean Score = {cv_results.mean():.4f} | Std Dev = {cv_results.std():.4f}"
        print(msg)
    
    # Plot model performance comparison
    fig = plt.figure(figsize=(10,6))
    fig.suptitle(f'Model Performance Comparison ({scoring})')
    ax = fig.add_subplot(111)
    plt.boxplot(results)
    ax.set_xticklabels(names)
    plt.show()
    return results, names

def train_and_save_best_model(X_train, y_train, X_val, y_val, models, preprocessor, task_type, save_path='best_universal_model.pkl'):
    """Train the top-performing model and save the full preprocessing+model pipeline."""
    # Get cross-validation results to find the best model
    results, names = evaluate_models(X_train, y_train, models, preprocessor, scoring='accuracy' if task_type == 'classification' else 'r2')
    best_model_idx = results.index(max(results))
    best_model_name = names[best_model_idx]
    best_model = next(model for name, model in models if name == best_model_name)
    
    # Build full pipeline (preprocessing + model)
    full_pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', best_model)])
    full_pipeline.fit(X_train, y_train)
    
    # Evaluate on validation set
    predictions = full_pipeline.predict(X_val)
    print("\n--- Best Model Validation Results ---")
    if task_type == 'classification':
        print(f"Accuracy: {accuracy_score(y_val, predictions):.4f}")
        print("\nClassification Report:\n", classification_report(y_val, predictions))
        print("Confusion Matrix:\n", confusion_matrix(y_val, predictions))
    elif task_type == 'regression':
        print(f"RMSE: {mean_squared_error(y_val, predictions, squared=False):.4f}")
        print(f"R2 Score: {r2_score(y_val, predictions):.4f}")
    
    # Save the entire pipeline (critical for handling raw data later)
    dump(full_pipeline, save_path)
    print(f"\nModel pipeline saved to {save_path}")
    return full_pipeline

# ------------------------------
# Example: Test with Iris Dataset
# ------------------------------
if __name__ == "__main__":
    # Configure these variables for YOUR dataset!
    DATA_PATH = "iris.data"
    TARGET_COL = 'class'
    FEATURE_COLS = ['sepal-length', 'sepal-width', 'petal-length', 'petal-width']
    TASK_TYPE = 'classification'
    SAVE_PATH = 'iris_universal_model.pkl'
    
    # Run the full pipeline
    X, y, preprocessor = load_and_preprocess_data(DATA_PATH, TARGET_COL, FEATURE_COLS)
    X_train, X_val, y_train, y_val = split_dataset(X, y)
    models = get_model_list(TASK_TYPE)
    train_and_save_best_model(X_train, y_train, X_val, y_val, models, preprocessor, TASK_TYPE, SAVE_PATH)

Step 2: Key Improvements for Universality

  • Modular, Parameterized Functions: Every step (data loading, splitting, evaluation) accepts parameters, so you can swap datasets or tasks in seconds.
  • Automatic Preprocessing: Handles numerical scaling and categorical encoding out of the box—no more manual data cleaning for new datasets.
  • Task-Agnostic: Supports both classification and regression; just set TASK_TYPE and the model list/scoring adjusts automatically.
  • Full Pipeline Saving: Saves preprocessing logic alongside the model, so when you load it later, it can process raw, unmodified data directly.
  • Extensible: Add your existing MLP (or any custom model) to the get_model_list function with one line of code.

Step 3: Use the Saved Model on New Data

To make predictions with your saved universal model:

loaded_pipeline = load('best_universal_model.pkl')

# Example: Predict on raw, unprocessed data
new_sample = pd.DataFrame({
    'sepal-length': [5.1],
    'sepal-width': [3.5],
    'petal-length': [1.4],
    'petal-width': [0.2]
})

prediction = loaded_pipeline.predict(new_sample)
print(f"Predicted class: {prediction[0]}")

内容的提问来源于stack exchange,提问作者pippo pluto

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.15 04:07:00