Python中为已保存CNN模型的音频输入生成GradCAM可视化图像方法
实现步骤与代码
你可以基于TensorFlow的GradientTape实现GradCAM可视化,最终生成的热图可以直接叠加在原始MFCC特征图上,直观展示哪部分音频特征对模型分类结果影响最大。
第一步:导入依赖
import tensorflow as tf import numpy as np import matplotlib.pyplot as plt import librosa.display import cv2
如果运行时报错缺少cv2依赖,先执行pip install opencv-python安装即可。
第二步:实现GradCAM热图生成函数
def generate_gradcam(input_mfcc, model, last_conv_layer_name="conv2d_3"): # 可先运行model.summary()确认最后一个卷积层的命名,默认4个Conv2D层的最后一个名为conv2d_3 grad_model = tf.keras.models.Model( inputs=[model.inputs], outputs=[model.get_layer(last_conv_layer_name).output, model.output] ) with tf.GradientTape() as tape: last_conv_layer_output, preds = grad_model(input_mfcc) pred_index = tf.argmax(preds[0]) target_class_output = preds[:, pred_index] # 计算目标类别输出相对于最后一层卷积特征的梯度 grads = tape.gradient(target_class_output, last_conv_layer_output) # 对梯度做全局平均池化得到每个滤波器的权重 pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2)) # 权重和卷积特征加权求和得到热图 last_conv_layer_output = last_conv_layer_output[0] heatmap = last_conv_layer_output @ pooled_grads[..., tf.newaxis] heatmap = tf.squeeze(heatmap) # 热图归一化到0-1区间 heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap) return heatmap.numpy(), pred_index
第三步:实现热图与MFCC叠加绘图函数
def plot_audio_gradcam(audio_path, model, label_encoder, nmfcc, max_pad_len, num_rows, num_columns, num_channels): # 提取音频MFCC特征 mfcc = extract_features(audio_path) input_mfcc = mfcc.reshape(1, num_rows, num_columns, num_channels) # 生成GradCAM热图和预测类别 heatmap, pred_index = generate_gradcam(input_mfcc, model) predicted_class = label_encoder.inverse_transform([pred_index])[0] # 热图resize到和MFCC特征相同尺寸 heatmap = np.uint8(255 * heatmap) heatmap = cv2.resize(heatmap, (mfcc.shape[1], mfcc.shape[0])) heatmap = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET) heatmap = cv2.cvtColor(heatmap, cv2.COLOR_BGR2RGB) heatmap = heatmap / 255.0 # 绘制原始MFCC和叠加热图的对比图 sample_rate = librosa.load(audio_path, sr=None)[1] fig, (ax_raw, ax_gradcam) = plt.subplots(2, 1, figsize=(12, 8)) # 原始MFCC raw_spec = librosa.display.specshow(mfcc, sr=sample_rate, x_axis='time', ax=ax_raw) ax_raw.set_title(f"原始MFCC特征,预测类别:{predicted_class}") fig.colorbar(raw_spec, ax=ax_raw) # 叠加GradCAM的MFCC gradcam_spec = librosa.display.specshow(mfcc, sr=sample_rate, x_axis='time', ax=ax_gradcam) ax_gradcam.imshow(heatmap, alpha=0.4, extent=raw_spec.get_extent()) ax_gradcam.set_title("GradCAM热图叠加(高亮区域为分类关键贡献区域)") fig.colorbar(gradcam_spec, ax=ax_gradcam) plt.tight_layout() plt.show()
调用方式
加载好训练完成的模型和LabelEncoder后,直接调用函数即可:
# 示例调用 plot_audio_gradcam( audio_path="测试音频.wav", model=加载好的模型实例, label_encoder=le, nmfcc=你的nmfcc参数值, max_pad_len=你的max_pad_len参数值, num_rows=你的num_rows参数值, num_columns=你的num_columns参数值, num_channels=你的num_channels参数值 )
内容的提问来源于stack exchange,提问作者Joe
相关产品推荐
相关产品推荐

