You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在tf.keras Dense层中设置常量权重(部分权重固定为0)

问题分析与解决方案

你遇到的核心问题是对tf.constant和模型权重的交互逻辑理解有误:当你把tf.constant(0)放进numpy数组再通过model.set_weights()赋值时,numpy无法识别TensorFlow张量,会自动把tf.constant(0)转换成普通的Python数值0。最终这个位置的权重依然是一个可训练的tf.Variable,反向传播时自然会计算它的梯度,这就是为什么梯度不为0的原因。

下面提供两种可行的解决思路,分别对应不同的场景需求:


方案一:自定义部分固定权重的Dense层(推荐)

这种方法从层的定义上区分可训练权重和固定常量,是最规范的实现方式,能从根源上让固定部分不参与梯度计算。

import tensorflow as tf
import numpy as np
tf.enable_eager_execution()

class PartialFixedDense(tf.keras.layers.Layer):
    def __init__(self, units, fixed_mask, fixed_values, activation=None, **kwargs):
        super().__init__(**kwargs)
        self.units = units
        # 掩码:True表示该位置权重固定,False表示可训练
        self.fixed_mask = tf.convert_to_tensor(fixed_mask, dtype=tf.bool)
        # 固定的权重值(仅掩码为True的位置生效)
        self.fixed_values = tf.convert_to_tensor(fixed_values, dtype=tf.float32)
        self.activation = tf.keras.activations.get(activation)

    def build(self, input_shape):
        input_dim = input_shape[-1]
        # 仅初始化可训练的权重(掩码为False的位置)
        self.trainable_weights_var = self.add_weight(
            shape=(input_dim, self.units),
            initializer='glorot_uniform',
            trainable=True,
            name='trainable_weights'
        )
        # 初始化偏置(默认全可训练,你也可以同理添加固定偏置的逻辑)
        self.bias = self.add_weight(
            shape=(self.units,),
            initializer='zeros',
            trainable=True,
            name='bias'
        )
        super().build(input_shape)

    def call(self, inputs):
        # 合并固定权重与可训练权重:掩码为True的位置用固定值,否则用可训练变量
        full_weights = tf.where(
            self.fixed_mask,
            self.fixed_values,
            self.trainable_weights_var
        )
        outputs = tf.matmul(inputs, full_weights) + self.bias
        if self.activation is not None:
            outputs = self.activation(outputs)
        return outputs

# 定义第一层的固定规则:固定(0,0)位置为0
fixed_mask = [[True, False], [False, False]]  # True表示固定
fixed_vals = [[0.0, 0.0], [0.0, 0.0]]        # 对应位置的固定值

# 构建模型
model = tf.keras.Sequential([
    PartialFixedDense(2, fixed_mask, fixed_vals, activation=tf.sigmoid, input_shape=(2,)),
    tf.keras.layers.Dense(2, activation=tf.sigmoid)
])

# 手动设置权重(和你原代码的权重对应)
model.layers[0].trainable_weights_var.assign(np.array([[0.0, 0.25],[0.2,0.3]]))
model.layers[0].bias.assign(np.array([0.35, 0.35]))
model.layers[1].set_weights([np.array([[0.4,0.5],[0.45, 0.55]]),np.array([0.6,0.6])])

# 损失与梯度函数保持不变
def loss(model, x, y):
    y_ = model(x)
    return tf.losses.mean_squared_error(labels=y, predictions=y_)

def grad(model, inputs, targets):
    with tf.GradientTape() as tape:
        loss_value = loss(model, inputs, targets)
    return loss_value, tape.gradient(loss_value, model.trainable_variables)

# 测试梯度:第一层可训练权重的梯度中不会包含固定位置的项
x = tf.random.normal((1, 2))
y = tf.random.normal((1, 2))
loss_val, grads = grad(model, x, y)
print("第一层可训练权重的梯度:\n", grads[0].numpy())

方案二:梯度更新时手动置零(快速验证)

如果你不想自定义层,也可以在计算梯度后,手动将固定位置的梯度强制设为0,这样更新时该位置权重不会改变。这种方法更快捷,但本质上权重还是可训练变量,只是被强制不更新。

import tensorflow as tf
import numpy as np
tf.enable_eager_execution()

model = tf.keras.Sequential([
    tf.keras.layers.Dense(2, activation=tf.sigmoid, input_shape=(2,)),
    tf.keras.layers.Dense(2, activation=tf.sigmoid)
])

# 直接用数值0初始化固定位置的权重,无需tf.constant
weights=[np.array([[0, 0.25],[0.2,0.3]]),np.array([0.35,0.35]),np.array([[0.4,0.5],[0.45, 0.55]]),np.array([0.6,0.6])]
model.set_weights(weights)

def loss(model, x, y):
    y_ = model(x)
    return tf.losses.mean_squared_error(labels=y, predictions=y_)

def grad(model, inputs, targets):
    with tf.GradientTape() as tape:
        loss_value = loss(model, inputs, targets)
    loss_value, grads = loss_value, tape.gradient(loss_value, model.trainable_variables)
    
    # 手动将第一层权重的(0,0)位置梯度置为0
    grads[0] = tf.tensor_scatter_nd_update(
        grads[0],
        indices=[[0, 0]],  # 需要置零的位置坐标
        updates=[0.0]
    )
    return loss_value, grads

# 测试梯度:(0,0)位置的梯度会被强制设为0
x = tf.random.normal((1, 2))
y = tf.random.normal((1, 2))
loss_val, grads = grad(model, x, y)
print("第一层权重的梯度:\n", grads[0].numpy())

内容的提问来源于stack exchange,提问作者Ev4

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.11 08:58:33