Python低依赖检测长音频中短片段时间位置的最简方法
极简Python实现方案(仅2个第三方依赖)
方案适配场景
针对Ableton Live无法导出带变速信息MIDI的问题,该方案可直接识别渲染得到的长节拍器WAV文件,区分不同音色的重拍、普通拍、分拍采样,自动生成带BPM变化、拍号变化的MIDI文件,输出结果可直接用于Clone Hero谱面制作。
- 依赖仅包含
numpy(轻量数值计算)、mido(纯Python MIDI处理),WAV读取直接使用Python标准库wave,无librosa、pytorch等重型音频/AI依赖 - 逻辑全程走时域匹配,无复杂频谱变换,代码量极小,可按需修改
- 相比零交叉检测方案,可精准区分不同音色的节拍点,对轻微底噪、音量波动的鲁棒性更强
核心实现逻辑
- 音频读取:统一将长节拍器音频、所有单节拍采样(小节重拍、四分音符拍、八分音符分拍)转为单声道归一化numpy数组,对齐采样率
- 节拍匹配:用滑动窗口余弦相似度匹配,遍历长音频查找所有和短采样音色一致的片段,记录对应时间点和节拍类型,匹配过的区间直接跳过避免重复检测
- 信息计算:将所有检测到的节拍点按时间排序,相邻四分音符/重拍的时间差计算瞬时BPM,相邻两个重拍之间的四分音符数量自动识别拍号
- MIDI导出:将计算得到的BPM转成MIDI标准tempo事件、拍号转成对应元事件,按时间顺序写入MIDI轨道即可
最小可运行代码
import wave import numpy as np import mido from mido import MidiFile, MidiTrack, MetaMessage def read_wav(file_path): """读取WAV文件为单声道归一化numpy数组,返回(音频数组, 采样率)""" with wave.open(file_path, "rb") as f: samp_rate = f.getframerate() total_frames = f.getnframes() raw_audio = np.frombuffer(f.readframes(total_frames), dtype=np.int16).astype(np.float32) # 转单声道 if f.getnchannels() == 2: raw_audio = raw_audio.reshape(-1, 2).mean(axis=1) # 幅值归一化 return raw_audio / np.max(np.abs(raw_audio)), samp_rate def detect_onsets(long_audio, target_sample, samp_rate, threshold=0.9): """在长音频中匹配所有目标采样的出现位置,返回时间点列表(单位:秒)""" onset_times = [] sample_len = len(target_sample) cursor = 0 step = int(samp_rate * 0.01) # 每次滑动10ms,平衡检测速度和精度 while cursor < len(long_audio) - sample_len: window = long_audio[cursor:cursor+sample_len] # 计算窗口和目标采样的余弦相似度 similarity = np.dot(window, target_sample) / (np.linalg.norm(window) * np.linalg.norm(target_sample) + 1e-8) if similarity > threshold: onset_times.append(cursor / samp_rate) cursor += sample_len # 跳过已匹配的采样长度,避免重复检测 else: cursor += step return onset_times if __name__ == "__main__": # 1. 读取所有音频文件 full_metronome_audio, sr = read_wav("full_metronome.wav") downbeat_sample, _ = read_wav("downbeat.wav") # 小节起始重拍采样 beat_sample, _ = read_wav("quarter_beat.wav") # 四分音符节拍采样 eighth_sample, _ = read_wav("eighth_beat.wav") # 八分音符分拍采样 # 2. 检测所有类型的节拍点 downbeat_times = detect_onsets(full_metronome_audio, downbeat_sample, sr) beat_times = detect_onsets(full_metronome_audio, beat_sample, sr) eighth_times = detect_onsets(full_metronome_audio, eighth_sample, sr) # 合并所有事件:格式为(时间秒, 类型) 类型0=重拍,1=四分拍,2=八分拍 all_events = sorted( [(t, 0) for t in downbeat_times] + [(t, 1) for t in beat_times] + [(t, 2) for t in eighth_times] ) # 3. 生成带速度、拍号信息的MIDI文件 midi_file = MidiFile(ticks_per_beat=480) tempo_track = MidiTrack() midi_file.tracks.append(tempo_track) last_event_time = 0 last_beat_time = None beat_count_after_downbeat = 0 default_tempo = mido.bpm2tempo(120) for event_time, event_type in all_events: time_delta = event_time - last_event_time tick_delta = int(mido.second2tick(time_delta, 480, default_tempo)) # 检测到重拍时写入拍号事件 if event_type == 0: time_sig_num = beat_count_after_downbeat + 1 tempo_track.append(MetaMessage( "time_signature", numerator=time_sig_num, denominator=4, time=tick_delta )) tick_delta = 0 beat_count_after_downbeat = 0 # 检测到四分拍/重拍时计算瞬时BPM,写入速度事件 if event_type in (0, 1): if last_beat_time is not None: instant_bpm = 60 / (event_time - last_beat_time) current_tempo = mido.bpm2tempo(instant_bpm) tempo_track.append(MetaMessage( "set_tempo", tempo=current_tempo, time=tick_delta )) tick_delta = 0 last_beat_time = event_time beat_count_after_downbeat += 1 last_event_time = event_time midi_file.save("clone_hero_tempo_map.mid")
使用说明
- 依赖安装仅需执行命令:
pip install numpy mido - 实际使用时可根据采样清晰度微调
detect_onsets函数里的threshold参数,取值范围0.8~0.95即可覆盖绝大多数场景,值越高匹配越严格 - 10分钟时长的音频在普通电脑上处理耗时不超过5秒,无内存压力
内容的提问来源于stack exchange,提问作者Swift142
相关产品推荐
相关产品推荐

