基于Librosa的donateacry-corpus音频增强脚本time_stretch函数TypeError问题
解决Librosa time_stretch函数TypeError错误及代码修复
错误原因分析
- Librosa API版本差异:新版本的
librosa.effects.time_stretch不再直接接收原始音频数据和拉伸速率作为位置参数,需要先计算音频的短时傅里叶变换(STFT)幅度谱,对幅度谱完成拉伸后再逆变换恢复音频。 - 自定义函数变量错误:
stretch方法定义的参数是y,但函数内部错误引用了未定义的data变量,导致逻辑混乱。
修复后的完整代码
import wave import librosa import numpy as np import matplotlib.pyplot as plt import os import glob import scipy import soundfile as sf class AudioAugmentation: def read_audio_file(self, file_path): data, sample_rate = sf.read(file_path) return data, sample_rate def write_audio_file(self, file, data, sample_rate): sf.write(file, data, sample_rate) def add_noise(self, data, noise_factor): noise = np.random.randn(len(data)) augmented_data = data + noise_factor * noise # 转换回原数据类型 augmented_data = augmented_data.astype(type(data[0])) return augmented_data def shift(self, data): y_shift = data.copy() timeshift_fac = 0.5 * 2 * (np.random.uniform() - 0.5) # 最大偏移长度的50% print("timeshift_fac = ", timeshift_fac) start = int(y_shift.shape[0] * timeshift_fac) print(start) if start > 0: y = np.pad(y_shift, (start, 0), mode='constant')[0:y_shift.shape[0]] else: y = np.pad(y_shift, (0, -start), mode='constant')[0:y_shift.shape[0]] return y.T def stretch(self, y, rate=1.0): # 修复:使用参数y而非未定义的data input_length = len(y) streching = y.copy().astype('float') # 适配Librosa新版本time_stretch用法 stft = librosa.stft(streching) stft_stretched = librosa.effects.time_stretch(stft, rate=rate) streching = librosa.istft(stft_stretched) # 保持原音频长度 if len(streching) > input_length: streching = streching[:input_length] else: streching = np.pad(streching, (0, max(0, input_length - len(streching))), "constant") return streching.astype(type(y[0])) # 实例化音频增强类 aa = AudioAugmentation() # 读取并生成增强音频 audio_dir = "/Users/kphil/Documents/UNUD/Capstone/donateacry_corpus_dataset/tired" output_dir = "/Users/kphil/Documents/UNUD/Capstone/donateacry_corpus_dataset/output/tired" # 确保输出目录存在 os.makedirs(output_dir, exist_ok=True) list1 = os.listdir(audio_dir) print(list1) for file in list1: if not file.startswith('.'): print(file) data, sr = aa.read_audio_file(os.path.join(audio_dir, file)) # 添加噪声 data_noise = aa.add_noise(data, 0.005) # 时移 data_roll = aa.shift(data) # 时间拉伸 data_stretch = aa.stretch(data, 1.1) # 保存增强后的音频 aa.write_audio_file(os.path.join(output_dir, f'generated1_{file}'), data_noise, sr) aa.write_audio_file(os.path.join(output_dir, f'generated2_{file}'), data_roll, sr) aa.write_audio_file(os.path.join(output_dir, f'generated3_{file}'), data_stretch, sr)
额外优化点
- 使用
os.path.join拼接路径,避免操作系统路径分隔符差异导致的错误 - 添加
os.makedirs(output_dir, exist_ok=True)确保输出目录存在,防止写入失败 - 拉伸后将音频数据转换回原数据类型,避免写入音频文件时出现格式问题
内容的提问来源于stack exchange,提问作者Kevin Philip
相关产品推荐
相关产品推荐

