如何用Python为批量音频(如肺音)实现带通滤波降噪?
批量音频带通滤波降噪与文本关联实现方案
针对你的肺音信号预处理需求,以下是具体的实现步骤和代码示例,全程基于Python完成:
1. 音频与文本文件的配对
首先要确保每个音频文件能精准匹配到对应的文本文件,假设你的文件命名规则是同名不同后缀(比如lung_sound_001.wav对应lung_sound_001.text),可以通过遍历文件夹自动配对:
import os # 定义文件夹路径 audio_dir = "./raw_audio" text_dir = "./text_files" processed_audio_dir = "./processed_audio" # 创建处理后音频的保存目录 os.makedirs(processed_audio_dir, exist_ok=True) # 遍历所有音频文件 audio_extensions = [".wav", ".flac", ".mp3"] # 根据你的音频格式调整 audio_files = [f for f in os.listdir(audio_dir) if os.path.splitext(f)[1].lower() in audio_extensions] # 建立音频-文本配对字典 audio_text_pairs = {} for audio_file in audio_files: base_name = os.path.splitext(audio_file)[0] text_file = f"{base_name}.text" text_path = os.path.join(text_dir, text_file) if os.path.exists(text_path): audio_text_pairs[os.path.join(audio_dir, audio_file)] = text_path else: print(f"警告:未找到{audio_file}对应的文本文件")
2. 带通滤波降噪实现
肺音的有效频率范围通常在50Hz-1000Hz,这里用Butterworth带通滤波器实现降噪,依赖scipy处理音频:
import numpy as np from scipy.io import wavfile from scipy.signal import butter, filtfilt def butter_bandpass(lowcut, highcut, fs, order=5): nyq = 0.5 * fs low = lowcut / nyq high = highcut / nyq b, a = butter(order, [low, high], btype='band') return b, a def bandpass_filter(data, lowcut, highcut, fs, order=5): b, a = butter_bandpass(lowcut, highcut, fs, order=order) y = filtfilt(b, a, data) return y # 处理单个音频文件 def process_audio(audio_path, save_path, lowcut=50, highcut=1000): fs, data = wavfile.read(audio_path) # 如果是多通道音频,转单通道(取第一通道) if len(data.shape) > 1: data = data[:, 0] # 归一化音频数据 data = data / np.max(np.abs(data)) # 应用带通滤波 filtered_data = bandpass_filter(data, lowcut, highcut, fs) # 转换为原数据类型(避免保存时格式错误) filtered_data = np.int16(filtered_data * 32767) # 保存处理后的音频 wavfile.write(save_path, fs, filtered_data)
3. 批量处理与关联维护
遍历所有配对好的音频-文本对,批量处理并保存,同时保证处理后的音频和原文本文件的关联关系:
# 批量处理所有音频 for audio_path, text_path in audio_text_pairs.items(): base_name = os.path.basename(audio_path) save_path = os.path.join(processed_audio_dir, base_name) process_audio(audio_path, save_path) print(f"已处理:{base_name},对应文本:{os.path.basename(text_path)}") # 可选:生成配对记录文件,方便后续深度学习加载 with open("./audio_text_mapping.csv", "w", encoding="utf-8") as f: f.write("processed_audio_path,text_path\n") for audio_path, text_path in audio_text_pairs.items(): processed_path = os.path.join(processed_audio_dir, os.path.basename(audio_path)) f.write(f"{processed_path},{text_path}\n")
4. 集成到深度学习预处理管道
如果用PyTorch构建数据集,可以直接在自定义Dataset类中加载处理后的音频和对应文本:
import torch from torch.utils.data import Dataset class LungSoundDataset(Dataset): def __init__(self, mapping_csv, transform=None): self.mapping = [] with open(mapping_csv, "r", encoding="utf-8") as f: next(f) # 跳过表头 for line in f: audio_path, text_path = line.strip().split(",") self.mapping.append((audio_path, text_path)) self.transform = transform def __len__(self): return len(self.mapping) def __getitem__(self, idx): audio_path, text_path = self.mapping[idx] # 加载处理后的音频(这里可以根据需求转成梅尔频谱等特征) fs, data = wavfile.read(audio_path) data = data / np.max(np.abs(data)) if self.transform: data = self.transform(data) # 加载文本内容 with open(text_path, "r", encoding="utf-8") as f: text = f.read().strip() return torch.tensor(data, dtype=torch.float32), text
内容的提问来源于stack exchange,提问作者khabat berwari
相关产品推荐
相关产品推荐

