如何在Scikit中构建线性加性模型并实现适配中间模型的自定义内核
加性高斯过程模型的sklearn实现方案
自定义内核是符合设计逻辑的正确实现思路,你之前的代码报错是因为sklearn的内核必须符合指定的接口规范,不能直接传入普通函数。以下是完整可运行的实现:
import numpy as np import matplotlib.pyplot as plt from sklearn.gaussian_process import GaussianProcessRegressor from sklearn.gaussian_process.kernels import RBF, Kernel, ConstantKernel # 全局参数 omega = 10*np.pi beta = np.pi alpha = 1 phi = .5*np.pi # 真实函数 def f(t): return alpha*np.tanh(beta*t)*np.sin(omega*t + phi) # 自定义中间模型内核 class IntermediateModelKernel(Kernel): def __init__(self, omega=omega): self.omega = omega def _f_I(self, t): return t * np.sin(self.omega * t) def __call__(self, X, Y=None, eval_gradient=False): if Y is None: Y = X # 计算中间模型输出的外积作为内核值 f_X = self._f_I(X).reshape(-1, 1) f_Y = self._f_I(Y).reshape(-1, 1) K = f_X @ f_Y.T if eval_gradient: # 中间模型无训练超参数,梯度返回空数组 return K, np.empty((X.shape[0], Y.shape[0], 0)) return K def diag(self, X): # 返回内核对角线元素,用于加速计算 return np.square(self._f_I(X).flatten()) def is_stationary(self): # 该内核为非平稳内核 return False # 生成训练数据 t = np.linspace(0, 1, 1000) X_train = np.random.choice(t.reshape(-1), size=12).reshape(-1, 1) y_train = f(X_train) # 构建组合内核:常数系数*中间模型内核 + RBF内核 kernel = ConstantKernel(constant_value_bounds=(1e-3, 1e3)) * IntermediateModelKernel() + \ RBF(length_scale=0.5, length_scale_bounds=(0.01, 100.0)) # 训练GPR,增加重启次数避免局部最优 gpr = GaussianProcessRegressor(kernel=kernel, random_state=1, n_restarts_optimizer=10) gpr.fit(X_train, y_train) # 输出优化得到的参数 print(f"优化后中间模型系数c_p: {np.sqrt(gpr.kernel_.k1.k1.constant_value):.4f}") print(f"优化后RBF内核长度尺度: {gpr.kernel_.k2.length_scale:.4f}") # 结果可视化 fig, ax = plt.subplots(dpi=200) ax.plot(t, f(t), 'grey', label="真实函数") ax.plot(X_train, y_train, 'ko', label="训练数据") ax.plot(t, gpr.predict(t.reshape(-1, 1)), 'r--', label="加性GP模型预测") ax.plot(t, t*np.sin(omega*t) * np.sqrt(gpr.kernel_.k1.k1.constant_value), 'g:', label="中间模型项") ax.legend() plt.show()
实现要点说明
- 所有适配sklearn的自定义内核都必须继承
Kernel基类,核心实现三个方法:__call__返回内核矩阵、diag返回内核对角线元素、is_stationary返回内核是否平稳的标识 - 代码中的
ConstantKernel对应你公式中的c_p²,开平方后就是你需要的系数c_p - GP训练过程会自动优化所有超参数,包括中间模型的系数和RBF内核的长度尺度,完全符合你提到的线性加性模型设计需求
内容的提问来源于stack exchange,提问作者Tom McLean
相关产品推荐
相关产品推荐

