You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

LightGBM与XGBoost回归模型预测时间出现负结果及性能极差问题排查求助

时间预测任务中LightGBM与XGBoost模型效果极差的问题排查与解决思路

首先,咱们先理清楚你的问题:你用回归算法做时间预测,样本数据里time值的跨度极大(从4.9到3122),但用LightGBM和XGBoost训练后,预测结果不仅质量差,还出现负值,甚至R2低到离谱,这确实得好好拆解下问题。

你的样本数据

先把你的样本整理成清晰的表格:

sizechannelstime
334.980278
3164.972054
3644.899884
32565.499221
35125.599495
310245.936933
1635.221653
16165.994821
16646.648254
162567.176828
165128.1707
1610248.651496
6437.801533
64167.398248
64648.395648
6425617.49494
6451226.43354
64102449.55192
256312.36093
2561620.50781
2566446.49553
256256170.5452
256512333.8809
2561024675.9459
512322.44313
5121653.82643
51264164.3493
512256659.4345
5125121306.881
51210243122.403

LightGBM实现与结果

你的代码

x_train, x_val, y_train, y_val = train_test_split(X,Y, train_size=0.8)
print(f"Number of training examples {len(x_train)}")
print(f"Number fo testing examples {len(x_val)}")
regressor = lightgbm.LGBMRegressor()
regressor.fit(x_train,y_train)
train_pred = regressor.predict(x_train)
train_rmse = mean_squared_error(train_pred, y_train) ** 0.5
print(f"Train RMSE is {train_rmse}")
val_pred = regressor.predict(x_val)
val_rmse = mean_squared_error(val_pred, y_val)**0.5
print(f"Test RMSE is {val_rmse}")
R_squared = r2_score(val_pred,y_val)
print('R2',R_squared)

你的结果

Train RMSE is 5385.50, Test RMSE is 1245.1,R2 -2.9991290197894976e+31

XGBoost实现(含Optuna优化)与结果

你的代码

def optimize(trial,x,y,regressor):
    max_depth = trial.suggest_int("max_depth",3,10)
    n_estimators = trial.suggest_int("n_estimators",5000,10000)
    max_leaves= trial.suggest_int("max_leaves",1,10)
    learning_rate = trial.suggest_loguniform('learning_rate', 0.001, 0.1)
    colsample_bytree = trial.suggest_uniform('colsample_bytree', 0.0, 1.0)
    min_child_weight = trial.suggest_uniform('min_child_weight',1,3)
    subsample = trial.suggest_uniform('subsample', 0.5, 1)
    model = xgb.XGBRegressor(
        objective ='reg:squarederror',
        n_estimators=n_estimators,
        max_depth=max_depth,
        learning_rate=learning_rate,
        colsample_bytree=colsample_bytree,
        min_child_weight=min_child_weight,
        max_leaves=max_leaves,
        subsample = subsample
    )
    kf=model_selection.KFold(n_splits=5)
    error=[]
    for idx in kf.split(X=x , y=y):
        train_idx , test_idx= idx[0],idx[1]
        xtrain=x[train_idx]
        ytrain=y[train_idx]
        xtest=x[test_idx]
        ytest=y[test_idx]
        model.fit(xtrain,ytrain)
        y_pred = model.predict(xtest)
        fold_err = metrics.mean_squared_error(ytest,y_pred)
        error.append(np.sqrt(fold_err))
    return np.mean(error)

best_params={'max_depth': 9, 'n_estimators': 9242, 'max_leaves': 7, 'learning_rate': 0.0015809052065858954, 'colsample_bytree': 0.4908644884609704, 'min_child_weight': 2.3502876962874435, 'subsample': 0.5927926099148189}

def optimize_xgb(X,y):
    list_of_y = ["Target 1"]
    for i,m in zip(range(y.shape[1]),list_of_y):
        print("{} optimized Parameters on MSE Error".format(m))
        optimization_function = partial(optimize , x=X,y=y[:,i],regressor="random_forest")
        study = optuna.create_study(direction="minimize")
        study.optimize(optimization_function,n_trials=50)

optimize_xgb(X_train, y_train)

def modeling(X, y, optimize = "no", max_depth=50, n_estimators=3000, max_leaves=30, learning_rate=0.01, colsample_bytree=1.0, gamma=0.0001, min_child_weight=2, reg_lambda=0.0001):
    if optimize == "no":
        model = xgb.XGBRegressor(objective='reg:squarederror')
    else:
        model = xgb.XGBRegressor(objective='reg:squarederror', **best_params)
    if y.shape[1] ==1:
        model_xgb = model.fit(X, y)
        cv = RepeatedKFold(n_splits=5, n_repeats=3, random_state=1)
        scores = []
        for i in range(y.shape[1]):
            scores.append(np.abs(cross_val_score(model, X, y[:,i], scoring='neg_mean_squared_error', cv=cv, n_jobs=-1)))
            print('Mean MSE of the {} target : {} ({})'.format(i,scores[i].mean(), scores[i].std()) )
        return model_xgb

model_xgb = modeling(X_train,y_train, optimize="yes")
model_xgb.fit(X_train, y_train)
y_pred = model_xgb.predict(X_test)
MSE = mse(y_pred,y_test)
RMSE = np.sqrt(MSE)
print("TEST MSE",MSE)
R_squared = r2_score(y_pred,y_test)
print("RMSE: ", np.round(RMSE, 2))
print("R-Squared: ", np.round(R_squared, 2))

你的结果

TEST MSE 2653915.139388934,RMSE: 1629.08,R-Squared: -1.69


问题根源分析

咱们一步步来拆解:

  1. R2计算完全错误:这是最明显的问题!r2_score的参数顺序是真实值在前,预测值在后,你写的r2_score(val_pred,y_val)和r2_score(y_pred,y_test)完全搞反了,这直接导致R2变成荒谬的负数,先把这个改过来,才能看到真实的模型表现。

  2. 目标值分布极端,缺乏预处理:你的time值跨度超过600倍(从4.9到3122),这种幂次增长的分布会让树模型难以学习——树模型对极端值非常敏感,容易把大部分注意力放在大数值样本上,忽略小数值样本,甚至在预测时出现负值(因为树模型在超出训练数据范围的区域外推时,很容易产生不合理的结果)。

  3. 模型参数完全不适配小样本:你总共只有30个样本,但XGBoost优化后用了n_estimators=9242,这绝对会严重过拟合!小样本下,几百棵树就足够了,上万棵树会把训练数据的噪声完全记住,泛化能力极差。LightGBM用默认参数也一样,默认的树数量和深度对小样本来说太复杂了。

  4. 数据集划分不合理:30个样本用8:2划分,验证集只有6个样本,结果波动极大,完全不能反映模型的真实性能,应该用交叉验证(比如5折或10折)来评估。


解决方案与修正代码

1. 先修正R2计算

把所有r2_score(pred, true)改成r2_score(true, pred),比如:

# LightGBM中修正
R_squared = r2_score(y_val, val_pred)

# XGBoost中修正
R_squared = r2_score(y_test, y_pred)

2. 数据预处理:对数变换

对目标变量time做对数变换,把极端分布压缩成平缓的分布,训练后再用指数变换还原结果:

import numpy as np

# 目标变量对数变换
Y_log = np.log(Y)
# 特征也可以考虑对数变换(因为size和channels也是幂次增长)
X_log = np.log(X)

3. 修正LightGBM代码(适配小样本+预处理)

import lightgbm as lgb
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_squared_error, r2_score

# 假设X是(size, channels)的二维数组,Y是time的一维数组
X_log = np.log(X)
Y_log = np.log(Y)

x_train, x_val, y_train, y_val = train_test_split(X_log, Y_log, train_size=0.8, random_state=42)
print(f"Number of training examples {len(x_train)}")
print(f"Number of testing examples {len(x_val)}")

# 调整参数适配小样本
regressor = lgb.LGBMRegressor(
    n_estimators=100,
    learning_rate=0.2,
    max_depth=4,
    random_state=42
)

# 添加早停,防止过拟合
regressor.fit(
    x_train, y_train,
    eval_set=[(x_val, y_val)],
    early_stopping_rounds=10,
    verbose=False
)

# 预测并还原对数变换
train_pred_log = regressor.predict(x_train)
train_pred = np.exp(train_pred_log)
train_rmse = mean_squared_error(train_pred, np.exp(y_train)) ** 0.5
print(f"Train RMSE is {train_rmse:.2f}")

val_pred_log = regressor.predict(x_val)
val_pred = np.exp(val_pred_log)
val_rmse = mean_squared_error(val_pred, np.exp(y_val))**0.5
print(f"Test RMSE is {val_rmse:.2f}")

# 修正后的R2计算
R_squared = r2_score(np.exp(y_val), val_pred)
print('R2', R_squared)

4. 修正XGBoost代码(优化参数范围+预处理)

import xgboost as xgb
import optuna
from sklearn.model_selection import RepeatedKFold, cross_val_score
from sklearn.metrics import mean_squared_error, r2_score
from functools import partial

# 数据预处理
X_log = np.log(X)
Y_log = np.log(Y)

# 调整Optuna搜索范围,适配小样本
def optimize(trial, x, y):
    max_depth = trial.suggest_int("max_depth", 3, 6)
    n_estimators = trial.suggest_int("n_estimators", 100, 500)
    max_leaves = trial.suggest_int("max_leaves", 2, 8)
    learning_rate = trial.suggest_loguniform('learning_rate', 0.01, 0.3)
    colsample_bytree = trial.suggest_uniform('colsample_bytree', 0.6, 1.0)
    min_child_weight = trial.suggest_int("min_child_weight", 1, 3)
    subsample = trial.suggest_uniform('subsample', 0.7, 1.0)
    
    model = xgb.XGBRegressor(
        objective='reg:squarederror',
        n_estimators=n_estimators,
        max_depth=max_depth,
        learning_rate=learning_rate,
        colsample_bytree=colsample_bytree,
        min_child_weight=min_child_weight,
        max_leaves=max_leaves,
        subsample=subsample,
        random_state=42
    )
    
    # 用重复交叉验证提升稳定性
    kf = RepeatedKFold(n_splits=5, n_repeats=2, random_state=42)
    errors = []
    for train_idx, test_idx in kf.split(x, y):
        xtrain, ytrain = x[train_idx], y[train_idx]
        xtest, ytest = x[test_idx], y[test_idx]
        model.fit(xtrain, ytrain)
        y_pred_log = model.predict(xtest)
        y_pred = np.exp(y_pred_log)
        fold_err = mean_squared_error(np.exp(ytest), y_pred)
        errors.append(np.sqrt(fold_err))
    
    return np.mean(errors)

# 优化参数(减少trial次数,小样本不需要太多)
study = optuna.create_study(direction="minimize", random_state=42)
study.optimize(lambda trial: optimize(trial, X_log, Y_log), n_trials=20)
best_params = study.best_params
print("Best parameters:", best_params)

# 建模与评估
model_xgb = xgb.XGBRegressor(objective='reg:squarederror', **best_params, random_state=42)
model_xgb.fit(X_log, Y_log)

# 交叉验证评估
cv = RepeatedKFold(n_splits=5, n_repeats=3, random_state=42)
scores = cross_val_score(
    model_xgb, X_log, Y_log,
    scoring='neg_mean_squared_error',
    cv=cv, n_jobs=-1
)
rmse_scores = np.sqrt(-scores)
print(f"Cross-Validation RMSE: {rmse_scores.mean():.2f} ± {rmse_scores.std():.2f}")

# 整体预测与评估
y_pred_log = model_xgb.predict(X_log)
y_pred = np.exp(y_pred_log)
r2 = r2_score(Y, y_pred)
print(f"Overall R2 Score: {r2:.2f}")

总结

核心问题就是R2计算顺序错误、目标值极端分布未处理、模型参数过拟合小样本,按照上面的步骤修正后,模型的预测效果应该会有显著提升,也不会再出现负值(对数变换后目标值都是正数,还原后自然也是正数)

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.30 02:57:32