You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Ray Tune优化Keras模型时遇ValueError:未返回指定mse指标

问题:Ray Tune优化Keras模型时出现Metric缺失错误

尝试用Ray Tune优化Keras模型(选择第一层最优隐藏层大小),但遇到以下错误:

ValueError: Trial returned a result which did not include the specified metric(s) mse that tune.TuneConfig() expects. Make sure your calls to tune.report() include the metric, or set the TUNE_DISABLE_STRICT_METRIC_CHECKING environment variable to 1. Result: {'trial_id': '830fd_00000', 'experiment_id': '3d78cf7f46b94e5390f528e95e97aff3', 'date': '2022-08-27_11-11-28', 'timestamp': 1661613088, 'pid': 1381126, 'hostname': 'arman-GT73EVR-7RE', 'node_ip': '172.30.50.84', 'done': True, 'config/threads': 8, 'config/lr': 0.055332536888805156, 'config/hidden': 65}

我的代码如下:

def train_Broad(config):
    import tensorflow as tf
    batch_size = 128
    epochs = 3
    np.random.seed(0)
    window_size = 200
    x_gyro, x_acc, x_mag, x_mag, q = load_data()

    x_gyro, x_acc, x_mag, q = shuffle(x_gyro, x_acc, x_mag, q)
    Att_quat = Att_q(q)
    x1 = Input((window_size, 3), name='x1')
    x2 = Input((window_size, 3), name='x2')
    convA1 = Conv1D(config["hidden"],11,padding='same',activation='relu')(x1)
    convA2 = Conv1D(10,11,padding='same',activation='relu')(convA1)
    poolA = MaxPooling1D(3)(convA2)
    convB1 = Conv1D(config["hidden"],11,padding='same',activation='relu')(x2)
    convB2 = Conv1D(10,11,padding='same',activation='relu')(convB1)
    poolB = MaxPooling1D(3)(convB2)
    AB = concatenate([poolA, poolB])
    lstm1 = Bidirectional(CuDNNGRU(10, return_sequences=True))(AB)
    drop1 = Dropout(0.25)(lstm1)
    lstm2 = Bidirectional(CuDNNGRU(10))(drop1)
    drop2 = Dropout(0.25)(lstm2)    
    y1_pred = Dense(4,kernel_regularizer='l2')(drop2)
    model = Model(inputs =[x1, x2], outputs = [y1_pred])
model.compile(
        loss="mse",
        optimizer=tf.keras.optimizers.Adam(learning_rate=config["lr"]), 
        metrics=["mse"])

    model.fit(
        [x_gyro, x_acc], 
        Att_quat,
        batch_size=batch_size,
        epochs=epochs,
        verbose=1,
        validation_split=0.1,
        callbacks=[keras.callbacks.EarlyStopping(monitor="mse", patience=1)])

def tune_Broad(num_training_iterations):
    tune.report(mean_loss="mse")
    sched = AsyncHyperBandScheduler(
        time_attr="training_iteration", max_t=400, grace_period=20
    )

    tuner = tune.Tuner(
        tune.with_resources(train_Broad, resources={"cpu": 4, "gpu": 1}),
        run_config=air.RunConfig(
            name="exp",
            stop={"mse": 0.0001, "training_iteration": num_training_iterations},
        ),
        tune_config=tune.TuneConfig(
            scheduler=sched,
            metric="mse",
            mode="min",
        ),
        
        param_space={
            "threads": 8,
            "lr": tune.uniform(0.001, 0.1),
            "hidden": tune.randint(1, 100),
        },
    )

    results = tuner.fit()
    print("Best hyperparameters found were: ", results.get_best_result().config)
    

if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument(
        "--smoke-test", action="store_true", help="Finish quickly for testing"
    )
    parser.add_argument(
        "--server-address",
        type=str,
        default=None,
        required=False,
        help="The address of server to connect to if using Ray Client.",
    )
    args, _ = parser.parse_known_args()
    if args.smoke_test:
        ray.init(num_cpus=4)
    elif args.server_address:
        ray.init(f"ray://{args.server_address}")

    tune_Broad(num_training_iterations=5 if args.smoke_test else 300)

解决方案

错误核心原因:TuneConfig指定追踪mse指标,但训练函数train_Broad未向Ray Tune报告该指标;同时tune_Broad中的tune.report(mean_loss="mse")完全无效(不在训练循环内,且传递的是字符串而非实际数值)。

需做以下修改:

1. 在训练函数中添加Tune回调报告指标

使用Ray Tune提供的TuneReportCallback,自动在每个epoch结束后向Tune报告指定指标,同时修正早停逻辑(建议用验证集指标避免过拟合):

from ray.tune.integration.keras import TuneReportCallback

# 修改model.fit的callbacks参数
model.fit(
    [x_gyro, x_acc], 
    Att_quat,
    batch_size=batch_size,
    epochs=epochs,
    verbose=1,
    validation_split=0.1,
    callbacks=[
        keras.callbacks.EarlyStopping(monitor="val_mse", patience=1),
        TuneReportCallback({"mse": "mse"})  # 映射Keras指标到Tune需要的字段
    ])

2. 删除无效的tune.report()调用

移除tune_Broad开头的tune.report(mean_loss="mse"),这行代码不在训练流程中,无法传递有效指标。

3. 修正代码缩进错误

原代码中model.compile()缩进错误,需与model = Model(...)同级:

model = Model(inputs =[x1, x2], outputs = [y1_pred])
model.compile(
    loss="mse",
    optimizer=tf.keras.optimizers.Adam(learning_rate=config["lr"]), 
    metrics=["mse"])

4. 补充必要模块导入

训练函数中需导入Keras相关层和模型类:

from tensorflow.keras.layers import Input, Conv1D, MaxPooling1D, concatenate, Bidirectional, CuDNNGRU, Dropout, Dense
from tensorflow.keras.models import Model

5. 修正数据加载的重复变量

原代码中x_gyro, x_acc, x_mag, x_mag, q = load_data()存在重复的x_mag变量,修正为:

x_gyro, x_acc, x_mag, _, q = load_data()  # 用下划线忽略多余的返回值

修改后的完整训练函数示例:

def train_Broad(config):
    import tensorflow as tf
    from tensorflow.keras.layers import Input, Conv1D, MaxPooling1D, concatenate, Bidirectional, CuDNNGRU, Dropout, Dense
    from tensorflow.keras.models import Model
    from tensorflow.keras import callbacks as keras_callbacks
    from ray.tune.integration.keras import TuneReportCallback
    import numpy as np
    from your_module import load_data, shuffle, Att_q  # 替换为实际模块名

    batch_size = 128
    epochs = 3
    np.random.seed(0)
    window_size = 200
    x_gyro, x_acc, x_mag, _, q = load_data()

    x_gyro, x_acc, x_mag, q = shuffle(x_gyro, x_acc, x_mag, q)
    Att_quat = Att_q(q)
    
    x1 = Input((window_size, 3), name='x1')
    x2 = Input((window_size, 3), name='x2')
    
    convA1 = Conv1D(config["hidden"], 11, padding='same', activation='relu')(x1)
    convA2 = Conv1D(10, 11, padding='same', activation='relu')(convA1)
    poolA = MaxPooling1D(3)(convA2)
    
    convB1 = Conv1D(config["hidden"], 11, padding='same', activation='relu')(x2)
    convB2 = Conv1D(10, 11, padding='same', activation='relu')(convB1)
    poolB = MaxPooling1D(3)(convB2)
    
    AB = concatenate([poolA, poolB])
    lstm1 = Bidirectional(CuDNNGRU(10, return_sequences=True))(AB)
    drop1 = Dropout(0.25)(lstm1)
    lstm2 = Bidirectional(CuDNNGRU(10))(drop1)
    drop2 = Dropout(0.25)(lstm2)    
    y1_pred = Dense(4, kernel_regularizer='l2')(drop2)
    
    model = Model(inputs=[x1, x2], outputs=[y1_pred])
    model.compile(
        loss="mse",
        optimizer=tf.keras.optimizers.Adam(learning_rate=config["lr"]), 
        metrics=["mse"])

    model.fit(
        [x_gyro, x_acc], 
        Att_quat,
        batch_size=batch_size,
        epochs=epochs,
        verbose=1,
        validation_split=0.1,
        callbacks=[
            keras_callbacks.EarlyStopping(monitor="val_mse", patience=1),
            TuneReportCallback({"mse": "mse"})
        ])

修改后的tune_Broad函数:

def tune_Broad(num_training_iterations):
    sched = AsyncHyperBandScheduler(
        time_attr="training_iteration", max_t=400, grace_period=20
    )

    tuner = tune.Tuner(
        tune.with_resources(train_Broad, resources={"cpu": 4, "gpu": 1}),
        run_config=air.RunConfig(
            name="exp",
            stop={"mse": 0.0001, "training_iteration": num_training_iterations},
        ),
        tune_config=tune.TuneConfig(
            scheduler=sched,
            metric="mse",
            mode="min",
        ),
        
        param_space={
            "threads": 8,
            "lr": tune.uniform(0.001, 0.1),
            "hidden": tune.randint(1, 100),
        },
    )

    results = tuner.fit()
    print("Best hyperparameters found were: ", results.get_best_result().config)

内容的提问来源于stack exchange,提问作者Arman Asgharpoor

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.21 06:24:16