如何自动定位两段人声音频中对应语句的词级起止时间?
我来分享几个实用的解决方案,搞定这个跨音频的单词时间戳映射问题~核心思路是利用语音内容匹配+时间规整技术,把A1中已知的单词分段精准映射到时长不同的A2音频上。下面分Python和C#两种主流方案详细说明:
Python 实现方案
快速落地:用Aeneas库一键对齐
Aeneas是专门做音频-文本对齐的开源工具,对多语言支持友好,完全适配我们的场景——既然A1和A2是同一句子,我们只需要把从A1 XML里提取的完整句子文本,拿去和A2音频做对齐,就能直接得到A2的单词时间戳,省心又高效。
步骤拆解:
- 从A1的XML文件里扒出所有单词,拼成完整句子文本(不用管A1的时间戳,Aeneas会重新对齐A2)
- 准备好A2的音频文件(支持WAV/MP3等常见格式)
- 调用Aeneas完成对齐,导出结果
- (可选)如果需要和A1的分段严格对应,可以结合DTW算法做帧级特征对齐,再映射时间戳
代码示例:
先装依赖:
pip install aeneas
然后写代码:
from aeneas.executetask import ExecuteTask from aeneas.task import Task import xml.etree.ElementTree as ET # 从A1的XML里提取句子文本(假设XML结构是<words><word text="xxx" start="0.0" end="1.2"/></words>) def get_sentence_from_a1_xml(xml_path): tree = ET.parse(xml_path) root = tree.getroot() words = [word.attrib["text"] for word in root.findall(".//word")] return " ".join(words) # 用Aeneas对齐A2音频和文本 def align_a2_to_text(audio_path, sentence, output_tsv): # 配置对齐任务:英语、纯文本、输出TSV格式 config = u"task_language=eng|is_text_type=plain|os_task_file_format=tsv" task = Task(config_string=config) task.audio_file_path_absolute = audio_path task.text_string = sentence task.sync_map_file_path_absolute = output_tsv # 执行对齐 ExecuteTask(task).execute() # 解析对齐结果 with open(output_tsv, "r", encoding="utf-8") as f: # 跳过表头行,读取每一行的时间戳和单词 rows = [line.strip().split("\t") for line in f.readlines()[1:]] return [{"text": row[3], "start": float(row[1]), "end": float(row[2])} for row in rows] # 主流程 if __name__ == "__main__": a1_xml = "a1_words.xml" a2_audio = "a2_audio.wav" output_file = "a2_word_timestamps.tsv" # 提取句子文本 target_sentence = get_sentence_from_a1_xml(a1_xml) print(f"待对齐的句子:{target_sentence}") # 对齐A2音频 a2_word_times = align_a2_to_text(a2_audio, target_sentence, output_file) # 打印结果 print("\nA2的单词时间戳:") for item in a2_word_times: print(f"单词:{item['text']} | 开始:{item['start']:.2f}s | 结束:{item['end']:.2f}s")
高精度进阶:DTW帧级对齐
如果觉得Aeneas的结果不够精细,可以用DTW(动态时间规整)算法,对A1和A2的音频特征(比如MFCC)做帧级对齐,再把A1的单词时间戳映射过去:
import librosa import numpy as np from dtw import dtw # 提取音频的MFCC特征(语音识别常用的特征) def extract_mfcc(audio_path): y, sr = librosa.load(audio_path, sr=None) mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13) return mfcc.T, sr # 返回帧级特征和采样率 # 用DTW对齐特征并映射时间戳 def map_a1_timestamps_to_a2(a1_audio, a2_audio, a1_word_times): # 提取特征 mfcc_a1, sr_a1 = extract_mfcc(a1_audio) mfcc_a2, sr_a2 = extract_mfcc(a2_audio) # 计算DTW对齐路径 _, _, _, path = dtw(mfcc_a1, mfcc_a2, dist=lambda x, y: np.linalg.norm(x - y, ord=1)) # 构建A1帧到A2帧的映射表 frame_map = {p[0]: p[1] for p in path} # 映射每个单词的时间戳 a2_word_times = [] for word in a1_word_times: # 把A1的时间转换为帧索引 a1_start_frame = int(word["start"] * sr_a1) a1_end_frame = int(word["end"] * sr_a1) # 找到对应的A2帧 a2_start_frame = frame_map.get(a1_start_frame, path[0][1]) a2_end_frame = frame_map.get(a1_end_frame, path[-1][1]) # 转换回时间 a2_start = a2_start_frame / sr_a2 a2_end = a2_end_frame / sr_a2 a2_word_times.append({ "text": word["text"], "start": a2_start, "end": a2_end }) return a2_word_times
(需要先装依赖:pip install librosa dtw-python)
C# 实现方案
自定义实现:NAudio + DTW
C#生态里没有像Aeneas那样开箱即用的对齐库,但可以用NAudio处理音频,自己实现DTW算法来完成对齐。
步骤拆解:
- 用NAudio读取A1和A2的音频,提取MFCC特征
- 实现DTW算法,计算两个特征序列的对齐路径
- 从A1的XML读取单词时间戳,通过对齐路径映射到A2
核心代码示例:
先装NAudio包:
Install-Package NAudio
然后写核心逻辑:
using NAudio.Wave; using NAudio.Dsp; using System; using System.Collections.Generic; using System.Xml.Linq; public class AudioWordAligner { // 提取音频的MFCC特征(简化版,可按需优化) public static double[][] ExtractMFCC(string audioPath, out int sampleRate) { using (var audioReader = new AudioFileReader(audioPath)) { sampleRate = audioReader.WaveFormat.SampleRate; var samples = new float[(int)audioReader.Length / (audioReader.WaveFormat.BitsPerSample / 8)]; audioReader.Read(samples, 0, samples.Length); int frameSize = 1024; int hopSize = 512; var mfccFrames = new List<double[]>(); // 分帧处理 for (int i = 0; i + frameSize < samples.Length; i += hopSize) { var frame = new float[frameSize]; Array.Copy(samples, i, frame, 0, frameSize); // 加汉明窗减少频谱泄漏 var window = WindowFunctions.Hamming(frameSize); for (int j = 0; j < frameSize; j++) { frame[j] *= window[j]; } // 计算FFT var fftComplex = new Complex[frameSize]; for (int j = 0; j < frameSize; j++) { fftComplex[j].X = frame[j]; fftComplex[j].Y = 0; } FastFourierTransform.FFT(true, (int)Math.Log(frameSize, 2), fftComplex); // 计算MFCC(这里是简化版,完整实现需要梅尔滤波器组等步骤) var mfcc = new double[13]; // 省略MFCC的详细计算,可参考开源实现或用Accord.NET库简化 mfccFrames.Add(mfcc); } return mfccFrames.ToArray(); } } // DTW算法实现,计算两个序列的对齐路径 public static int[][] ComputeDTWPath(double[][] seq1, double[][] seq2) { int n = seq1.Length; int m = seq2.Length; double[,] costMatrix = new double[n + 1, m + 1]; int[,] pathDirection = new int[n + 1, m + 1]; // 初始化代价矩阵 for (int i = 0; i <= n; i++) costMatrix[i, 0] = double.PositiveInfinity; for (int j = 0; j <= m; j++) costMatrix[0, j] = double.PositiveInfinity; costMatrix[0, 0] = 0; // 填充代价矩阵并记录路径方向 for (int i = 1; i <= n; i++) { for (int j = 1; j <= m; j++) { double distance = EuclideanDistance(seq1[i - 1], seq2[j - 1]); costMatrix[i, j] = distance + Math.Min(Math.Min(costMatrix[i - 1, j], costMatrix[i, j - 1]), costMatrix[i - 1, j - 1]); // 记录路径方向:1=上,2=左,3=对角线 if (costMatrix[i - 1, j] == Math.Min(Math.Min(costMatrix[i - 1, j], costMatrix[i, j - 1]), costMatrix[i - 1, j - 1])) pathDirection[i, j] = 1; else if (costMatrix[i, j - 1] == Math.Min(Math.Min(costMatrix[i - 1, j], costMatrix[i, j - 1]), costMatrix[i - 1, j - 1])) pathDirection[i, j] = 2; else pathDirection[i, j] = 3; } } // 回溯得到对齐路径 var path = new List<int[]>(); int x = n, y = m; while (x > 0 && y > 0) { path.Add(new int[] { x - 1, y - 1 }); if (pathDirection[x, y] == 1) x--; else if (pathDirection[x, y] == 2) y--; else { x--; y--; } } path.Reverse(); return path.ToArray(); } // 计算欧氏距离 private static double EuclideanDistance(double[] a, double[] b) { double sum = 0; for (int i = 0; i < a.Length; i++) { sum += Math.Pow(a[i] - b[i], 2); } return Math.Sqrt(sum); } // 从XML读取A1的单词时间戳 public static List<WordTime> LoadA1WordTimes(string xmlPath) { var doc = XDocument.Load(xmlPath); var wordTimes = new List<WordTime>(); foreach (var wordElem in doc.Descendants("word")) { wordTimes.Add(new WordTime { Text = wordElem.Attribute("text").Value, Start = double.Parse(wordElem.Attribute("start").Value), End = double.Parse(wordElem.Attribute("end").Value) }); } return wordTimes; } // 将A1的时间戳映射到A2 public static List<WordTime> MapToA2(string a1AudioPath, string a2AudioPath, List<WordTime> a1WordTimes) { var mfccA1 = ExtractMFCC(a1AudioPath, out int srA1); var mfccA2 = ExtractMFCC(a2AudioPath, out int srA2); var dtwPath = ComputeDTWPath(mfccA1, mfccA2); // 构建帧映射表 var frameMap = new Dictionary<int, int>(); foreach (var pair in dtwPath) { frameMap[pair[0]] = pair[1]; } var a2WordTimes = new List<WordTime>(); foreach (var word in a1WordTimes) { // 转换为帧索引(hopSize=512) int a1StartFrame = (int)(word.Start * srA1 / 512); int a1EndFrame = (int)(word.End * srA1 / 512); if (frameMap.TryGetValue(a1StartFrame, out int a2StartFrame) && frameMap.TryGetValue(a1EndFrame, out int a2EndFrame)) { a2WordTimes.Add(new WordTime { Text = word.Text, Start = a2StartFrame * 512.0 / srA2, End = a2EndFrame * 512.0 / srA2 }); } } return a2WordTimes; } // 存储单词时间戳的类 public class WordTime { public string Text { get; set; } public double Start { get; set; } public double End { get; set; } } } // 使用示例 class Program { static void Main(string[] args) { string a1Xml = "a1_words.xml"; string a1Audio = "a1_audio.wav"; string a2Audio = "a2_audio.wav"; var a1Times = AudioWordAligner.LoadA1WordTimes(a1Xml); var a2Times = AudioWordAligner.MapToA2(a1Audio, a2Audio, a1Times); Console.WriteLine("A2的单词时间戳:"); foreach (var word in a2Times) { Console.WriteLine($"单词:{word.Text} | 开始:{word.Start:F2}s | 结束:{word.End:F2}s"); } } }
小提示
- Python方案里,Aeneas适合快速落地,DTW适合需要高精度自定义的场景;
- C#方案里,MFCC的实现可以用
Accord.NET库简化,避免自己写复杂的特征提取逻辑; - 尽量用WAV格式的音频,减少格式转换带来的误差。
内容的提问来源于stack exchange,提问作者Kadir Şahbaz
相关产品推荐
相关产品推荐

