You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何自动定位两段人声音频中对应语句的词级起止时间?

我来分享几个实用的解决方案,搞定这个跨音频的单词时间戳映射问题~核心思路是利用语音内容匹配+时间规整技术,把A1中已知的单词分段精准映射到时长不同的A2音频上。下面分Python和C#两种主流方案详细说明:

Python 实现方案

快速落地:用Aeneas库一键对齐

Aeneas是专门做音频-文本对齐的开源工具,对多语言支持友好,完全适配我们的场景——既然A1和A2是同一句子,我们只需要把从A1 XML里提取的完整句子文本,拿去和A2音频做对齐,就能直接得到A2的单词时间戳,省心又高效。

步骤拆解:

  1. 从A1的XML文件里扒出所有单词,拼成完整句子文本(不用管A1的时间戳,Aeneas会重新对齐A2)
  2. 准备好A2的音频文件(支持WAV/MP3等常见格式)
  3. 调用Aeneas完成对齐,导出结果
  4. (可选)如果需要和A1的分段严格对应,可以结合DTW算法做帧级特征对齐,再映射时间戳

代码示例:

先装依赖:

pip install aeneas

然后写代码:

from aeneas.executetask import ExecuteTask
from aeneas.task import Task
import xml.etree.ElementTree as ET

# 从A1的XML里提取句子文本(假设XML结构是<words><word text="xxx" start="0.0" end="1.2"/></words>)
def get_sentence_from_a1_xml(xml_path):
    tree = ET.parse(xml_path)
    root = tree.getroot()
    words = [word.attrib["text"] for word in root.findall(".//word")]
    return " ".join(words)

# 用Aeneas对齐A2音频和文本
def align_a2_to_text(audio_path, sentence, output_tsv):
    # 配置对齐任务:英语、纯文本、输出TSV格式
    config = u"task_language=eng|is_text_type=plain|os_task_file_format=tsv"
    task = Task(config_string=config)
    task.audio_file_path_absolute = audio_path
    task.text_string = sentence
    task.sync_map_file_path_absolute = output_tsv

    # 执行对齐
    ExecuteTask(task).execute()

    # 解析对齐结果
    with open(output_tsv, "r", encoding="utf-8") as f:
        # 跳过表头行,读取每一行的时间戳和单词
        rows = [line.strip().split("\t") for line in f.readlines()[1:]]
        return [{"text": row[3], "start": float(row[1]), "end": float(row[2])} for row in rows]

# 主流程
if __name__ == "__main__":
    a1_xml = "a1_words.xml"
    a2_audio = "a2_audio.wav"
    output_file = "a2_word_timestamps.tsv"

    # 提取句子文本
    target_sentence = get_sentence_from_a1_xml(a1_xml)
    print(f"待对齐的句子:{target_sentence}")

    # 对齐A2音频
    a2_word_times = align_a2_to_text(a2_audio, target_sentence, output_file)

    # 打印结果
    print("\nA2的单词时间戳:")
    for item in a2_word_times:
        print(f"单词:{item['text']} | 开始:{item['start']:.2f}s | 结束:{item['end']:.2f}s")

高精度进阶:DTW帧级对齐

如果觉得Aeneas的结果不够精细,可以用DTW(动态时间规整)算法,对A1和A2的音频特征(比如MFCC)做帧级对齐,再把A1的单词时间戳映射过去:

import librosa
import numpy as np
from dtw import dtw

# 提取音频的MFCC特征(语音识别常用的特征)
def extract_mfcc(audio_path):
    y, sr = librosa.load(audio_path, sr=None)
    mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=13)
    return mfcc.T, sr  # 返回帧级特征和采样率

# 用DTW对齐特征并映射时间戳
def map_a1_timestamps_to_a2(a1_audio, a2_audio, a1_word_times):
    # 提取特征
    mfcc_a1, sr_a1 = extract_mfcc(a1_audio)
    mfcc_a2, sr_a2 = extract_mfcc(a2_audio)

    # 计算DTW对齐路径
    _, _, _, path = dtw(mfcc_a1, mfcc_a2, dist=lambda x, y: np.linalg.norm(x - y, ord=1))

    # 构建A1帧到A2帧的映射表
    frame_map = {p[0]: p[1] for p in path}

    # 映射每个单词的时间戳
    a2_word_times = []
    for word in a1_word_times:
        # 把A1的时间转换为帧索引
        a1_start_frame = int(word["start"] * sr_a1)
        a1_end_frame = int(word["end"] * sr_a1)

        # 找到对应的A2帧
        a2_start_frame = frame_map.get(a1_start_frame, path[0][1])
        a2_end_frame = frame_map.get(a1_end_frame, path[-1][1])

        # 转换回时间
        a2_start = a2_start_frame / sr_a2
        a2_end = a2_end_frame / sr_a2

        a2_word_times.append({
            "text": word["text"],
            "start": a2_start,
            "end": a2_end
        })
    return a2_word_times

(需要先装依赖:pip install librosa dtw-python)

C# 实现方案

自定义实现:NAudio + DTW

C#生态里没有像Aeneas那样开箱即用的对齐库,但可以用NAudio处理音频,自己实现DTW算法来完成对齐。

步骤拆解:

  1. 用NAudio读取A1和A2的音频,提取MFCC特征
  2. 实现DTW算法,计算两个特征序列的对齐路径
  3. 从A1的XML读取单词时间戳,通过对齐路径映射到A2

核心代码示例:

先装NAudio包:

Install-Package NAudio

然后写核心逻辑:

using NAudio.Wave;
using NAudio.Dsp;
using System;
using System.Collections.Generic;
using System.Xml.Linq;

public class AudioWordAligner
{
    // 提取音频的MFCC特征(简化版,可按需优化)
    public static double[][] ExtractMFCC(string audioPath, out int sampleRate)
    {
        using (var audioReader = new AudioFileReader(audioPath))
        {
            sampleRate = audioReader.WaveFormat.SampleRate;
            var samples = new float[(int)audioReader.Length / (audioReader.WaveFormat.BitsPerSample / 8)];
            audioReader.Read(samples, 0, samples.Length);

            int frameSize = 1024;
            int hopSize = 512;
            var mfccFrames = new List<double[]>();

            // 分帧处理
            for (int i = 0; i + frameSize < samples.Length; i += hopSize)
            {
                var frame = new float[frameSize];
                Array.Copy(samples, i, frame, 0, frameSize);

                // 加汉明窗减少频谱泄漏
                var window = WindowFunctions.Hamming(frameSize);
                for (int j = 0; j < frameSize; j++)
                {
                    frame[j] *= window[j];
                }

                // 计算FFT
                var fftComplex = new Complex[frameSize];
                for (int j = 0; j < frameSize; j++)
                {
                    fftComplex[j].X = frame[j];
                    fftComplex[j].Y = 0;
                }
                FastFourierTransform.FFT(true, (int)Math.Log(frameSize, 2), fftComplex);

                // 计算MFCC(这里是简化版,完整实现需要梅尔滤波器组等步骤)
                var mfcc = new double[13];
                // 省略MFCC的详细计算,可参考开源实现或用Accord.NET库简化
                mfccFrames.Add(mfcc);
            }
            return mfccFrames.ToArray();
        }
    }

    // DTW算法实现,计算两个序列的对齐路径
    public static int[][] ComputeDTWPath(double[][] seq1, double[][] seq2)
    {
        int n = seq1.Length;
        int m = seq2.Length;
        double[,] costMatrix = new double[n + 1, m + 1];
        int[,] pathDirection = new int[n + 1, m + 1];

        // 初始化代价矩阵
        for (int i = 0; i <= n; i++) costMatrix[i, 0] = double.PositiveInfinity;
        for (int j = 0; j <= m; j++) costMatrix[0, j] = double.PositiveInfinity;
        costMatrix[0, 0] = 0;

        // 填充代价矩阵并记录路径方向
        for (int i = 1; i <= n; i++)
        {
            for (int j = 1; j <= m; j++)
            {
                double distance = EuclideanDistance(seq1[i - 1], seq2[j - 1]);
                costMatrix[i, j] = distance + Math.Min(Math.Min(costMatrix[i - 1, j], costMatrix[i, j - 1]), costMatrix[i - 1, j - 1]);
                
                // 记录路径方向:1=上,2=左,3=对角线
                if (costMatrix[i - 1, j] == Math.Min(Math.Min(costMatrix[i - 1, j], costMatrix[i, j - 1]), costMatrix[i - 1, j - 1]))
                    pathDirection[i, j] = 1;
                else if (costMatrix[i, j - 1] == Math.Min(Math.Min(costMatrix[i - 1, j], costMatrix[i, j - 1]), costMatrix[i - 1, j - 1]))
                    pathDirection[i, j] = 2;
                else
                    pathDirection[i, j] = 3;
            }
        }

        // 回溯得到对齐路径
        var path = new List<int[]>();
        int x = n, y = m;
        while (x > 0 && y > 0)
        {
            path.Add(new int[] { x - 1, y - 1 });
            if (pathDirection[x, y] == 1)
                x--;
            else if (pathDirection[x, y] == 2)
                y--;
            else
            {
                x--;
                y--;
            }
        }
        path.Reverse();
        return path.ToArray();
    }

    // 计算欧氏距离
    private static double EuclideanDistance(double[] a, double[] b)
    {
        double sum = 0;
        for (int i = 0; i < a.Length; i++)
        {
            sum += Math.Pow(a[i] - b[i], 2);
        }
        return Math.Sqrt(sum);
    }

    // 从XML读取A1的单词时间戳
    public static List<WordTime> LoadA1WordTimes(string xmlPath)
    {
        var doc = XDocument.Load(xmlPath);
        var wordTimes = new List<WordTime>();
        foreach (var wordElem in doc.Descendants("word"))
        {
            wordTimes.Add(new WordTime
            {
                Text = wordElem.Attribute("text").Value,
                Start = double.Parse(wordElem.Attribute("start").Value),
                End = double.Parse(wordElem.Attribute("end").Value)
            });
        }
        return wordTimes;
    }

    // 将A1的时间戳映射到A2
    public static List<WordTime> MapToA2(string a1AudioPath, string a2AudioPath, List<WordTime> a1WordTimes)
    {
        var mfccA1 = ExtractMFCC(a1AudioPath, out int srA1);
        var mfccA2 = ExtractMFCC(a2AudioPath, out int srA2);
        var dtwPath = ComputeDTWPath(mfccA1, mfccA2);

        // 构建帧映射表
        var frameMap = new Dictionary<int, int>();
        foreach (var pair in dtwPath)
        {
            frameMap[pair[0]] = pair[1];
        }

        var a2WordTimes = new List<WordTime>();
        foreach (var word in a1WordTimes)
        {
            // 转换为帧索引(hopSize=512)
            int a1StartFrame = (int)(word.Start * srA1 / 512);
            int a1EndFrame = (int)(word.End * srA1 / 512);

            if (frameMap.TryGetValue(a1StartFrame, out int a2StartFrame) && frameMap.TryGetValue(a1EndFrame, out int a2EndFrame))
            {
                a2WordTimes.Add(new WordTime
                {
                    Text = word.Text,
                    Start = a2StartFrame * 512.0 / srA2,
                    End = a2EndFrame * 512.0 / srA2
                });
            }
        }
        return a2WordTimes;
    }

    // 存储单词时间戳的类
    public class WordTime
    {
        public string Text { get; set; }
        public double Start { get; set; }
        public double End { get; set; }
    }
}

// 使用示例
class Program
{
    static void Main(string[] args)
    {
        string a1Xml = "a1_words.xml";
        string a1Audio = "a1_audio.wav";
        string a2Audio = "a2_audio.wav";

        var a1Times = AudioWordAligner.LoadA1WordTimes(a1Xml);
        var a2Times = AudioWordAligner.MapToA2(a1Audio, a2Audio, a1Times);

        Console.WriteLine("A2的单词时间戳:");
        foreach (var word in a2Times)
        {
            Console.WriteLine($"单词:{word.Text} | 开始:{word.Start:F2}s | 结束:{word.End:F2}s");
        }
    }
}

小提示

  • Python方案里,Aeneas适合快速落地,DTW适合需要高精度自定义的场景;
  • C#方案里,MFCC的实现可以用Accord.NET库简化,避免自己写复杂的特征提取逻辑;
  • 尽量用WAV格式的音频,减少格式转换带来的误差。

内容的提问来源于stack exchange,提问作者Kadir Şahbaz

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.20 11:57:56