You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将Azure TTS生成的OGG数据写入WebRTC音频轨道?代码排查

问题描述

使用Azure Speech SDK实现文本转语音(TTS)生成OGG格式数据,流式输出发送至WebRTC音频轨道,数据能成功发送到对等端,但客户端听不到音频。相关代码如下:

发送音频数据的函数

func sendAudioData(data io.Reader, audioTrack *webrtc.TrackLocalStaticSample) error {
    ogg, _, oggErr := oggreader.NewWith(data)
    if oggErr != nil {
        return oggErr
    }
    var lastGranule uint64
    for {
        pageData, pageHeader, oggErr := ogg.ParseNextPage()
        if errors.Is(oggErr, io.EOF) {
            fmt.Printf("All audio pages parsed and sent")
            return nil
        }

        if oggErr != nil {
            return oggErr
        }
        slog.Info("ParseNextPage", "header", pageHeader)
        // The amount of samples is the difference between the last and current timestamp
        sampleCount := float64(pageHeader.GranulePosition - lastGranule)
        lastGranule = pageHeader.GranulePosition
        sampleDuration := time.Duration((sampleCount/48000)*1000) * time.Millisecond

        slog.Info("Write Sample", "duration", sampleDuration, "length", len(pageData))
        if oggErr = audioTrack.WriteSample(media.Sample{Data: pageData, Duration: sampleDuration}); oggErr != nil {
            return oggErr
        }
    }
}

生成流式数据的代码

package speech

import (
    "errors"
    "fmt"
    "os"
    "time"

    "github.com/Microsoft/cognitive-services-speech-sdk-go/audio"
    "github.com/Microsoft/cognitive-services-speech-sdk-go/common"
    "github.com/Microsoft/cognitive-services-speech-sdk-go/speech"
)

func azureTTSStream(text, lang string) (*speech.AudioDataStream, error) {
    // stream, err := audio.CreatePullAudioOutputStream()
    // if err != nil {
    //  return nil, err
    // }
    // audioConfig, err := audio.NewAudioConfigFromStreamOutput(stream)
    // if err != nil {
    //  panic(err)
    // }
    speechKey := os.Getenv("AZURE_SPEECH_KEY")
    speechRegion := os.Getenv("AZURE_SPEECH_REGION")
    speechConfig, err := speech.NewSpeechConfigFromSubscription(speechKey, speechRegion)
    if err != nil {
        panic(err)
    }
    speechConfig.SetSpeechSynthesisOutputFormat(common.Ogg48Khz16BitMonoOpus)
    synthesizer, err := speech.NewSpeechSynthesizerFromConfig(speechConfig, nil)
    if err != nil {
        panic(err)
    }

    voice := azureVoices[lang]
    input := fmt.Sprintf(azureSAMLInput, lang, voice, text)

    task := synthesizer.SpeakSsmlAsync(input)
    var outcome speech.SpeechSynthesisOutcome
    select {
    case outcome = <-task:
    case <-time.After(60 * time.Second):
        fmt.Println("timeout")
        return nil, errors.New("timeout")
    }
    defer outcome.Close()
    if outcome.Error != nil {
        fmt.Println("outcome error: ", outcome.Error)
    }
    
    return speech.NewAudioDataStreamFromSpeechSynthesisResult(outcome.Result)

    // if outcome.Result.Reason == common.SynthesizingAudioCompleted {
    //  fmt.Println("outcome completed")
    // }

    // cancellation, _ := speech.NewCancellationDetailsFromSpeechSynthesisResult(outcome.Result)
    // msg := fmt.Sprintf("CANCELED: Reason=%d.\n", cancellation.Reason)
    // if cancellation.Reason == common.Error {
    //  fmt.Sprintf("%s ErrorCode=%d, ErrorDetails=[%s]\n", msg, cancellation.ErrorCode, cancellation.ErrorDetails)
    // }
}

问题排查与修复方案

核心问题分析

  1. OGG容器数据直接写入错误:WebRTC音频轨道需要裸Opus编码帧,但你直接将包含OGG容器头的页面数据pageData写入轨道,客户端无法解析容器格式,自然听不到声音。
  2. 时长计算逻辑错误:Azure TTS输出的OGG Opus中,GranulePosition并非直接对应PCM样本数,用固定48000采样率计算时长会导致时间轴混乱,引发播放不同步或静音。
  3. 错误处理与资源管理缺失:原代码未检查TTS合成是否成功完成,也未正确关闭合成器资源,可能导致数据流不完整或格式损坏。

修复后的代码

1. 修正sendAudioData函数(提取裸Opus帧)

使用opusreader解析OGG页面中的纯Opus帧,再写入WebRTC轨道:

import (
    "io"
    "errors"
    "time"
    "github.com/pion/webrtc/v3"
    "github.com/pion/webrtc/v3/media"
    "github.com/pion/opus"
    "github.com/pion/oggreader"
)

func sendAudioData(data io.Reader, audioTrack *webrtc.TrackLocalStaticSample) error {
    ogg, _, err := oggreader.NewWith(data)
    if err != nil {
        return err
    }

    // 初始化Opus解码器,用于计算单帧时长
    opusDecoder, err := opus.NewDecoder(48000, 1)
    if err != nil {
        return err
    }

    for {
        pageData, pageHeader, err := ogg.ParseNextPage()
        if errors.Is(err, io.EOF) {
            return nil
        }
        if err != nil {
            return err
        }

        // 从OGG页面中提取所有裸Opus帧
        opusFrames, err := opusreader.ParseFrames(pageData)
        if err != nil {
            return err
        }

        for _, frame := range opusFrames {
            // 计算当前Opus帧的精确时长
            frameSamples, err := opusDecoder.FrameSamples(frame)
            if err != nil {
                return err
            }
            duration := time.Duration((float64(frameSamples)/48000)*1000) * time.Millisecond

            // 将裸Opus帧写入WebRTC轨道
            if err := audioTrack.WriteSample(media.Sample{
                Data:     frame,
                Duration: duration,
            }); err != nil {
                return err
            }
        }
    }
}

2. 修正azureTTSStream函数(完善错误处理与资源管理)

补充必要的参数校验、合成结果检查,确保数据流完整:

package speech

import (
    "errors"
    "fmt"
    "os"
    "time"

    "github.com/Microsoft/cognitive-services-speech-sdk-go/common"
    "github.com/Microsoft/cognitive-services-speech-sdk-go/speech"
)

func azureTTSStream(text, lang string) (*speech.AudioDataStream, error) {
    // 检查环境变量是否配置
    speechKey := os.Getenv("AZURE_SPEECH_KEY")
    speechRegion := os.Getenv("AZURE_SPEECH_REGION")
    if speechKey == "" || speechRegion == "" {
        return nil, errors.New("AZURE_SPEECH_KEY或AZURE_SPEECH_REGION环境变量未设置")
    }

    speechConfig, err := speech.NewSpeechConfigFromSubscription(speechKey, speechRegion)
    if err != nil {
        return nil, err
    }
    // 明确指定输出格式为OGG Opus 48kHz单声道
    speechConfig.SetSpeechSynthesisOutputFormat(common.Ogg48Khz16BitMonoOpus)

    synthesizer, err := speech.NewSpeechSynthesizerFromConfig(speechConfig, nil)
    if err != nil {
        return nil, err
    }
    defer synthesizer.Close() // 确保合成器资源被正确释放

    // 检查语言对应的语音是否存在
    voice, ok := azureVoices[lang]
    if !ok {
        return nil, fmt.Errorf("不支持的语言:%s", lang)
    }
    input := fmt.Sprintf(azureSAMLInput, lang, voice, text)

    task := synthesizer.SpeakSsmlAsync(input)
    var outcome speech.SpeechSynthesisOutcome
    select {
    case outcome = <-task:
    case <-time.After(60 * time.Second):
        return nil, errors.New("TTS合成超时")
    }
    defer outcome.Close()

    // 检查合成是否出错
    if outcome.Error != nil {
        return nil, fmt.Errorf("TTS合成失败:%w", outcome.Error)
    }

    // 确认合成已完成
    if outcome.Result.Reason != common.SynthesizingAudioCompleted {
        return nil, errors.New("TTS合成未正常完成")
    }

    return speech.NewAudioDataStreamFromSpeechSynthesisResult(outcome.Result), nil
}

额外注意事项

  • WebRTC轨道配置:创建音频轨道时必须指定Opus编解码器,且参数与Azure TTS输出匹配:
    audioTrack, err := webrtc.NewTrackLocalStaticSample(webrtc.RTPCodecCapability{MimeType: webrtc.MimeTypeOpus}, "audio", "pion")
    
  • 传输时机:确保WebRTC PeerConnection完成ICE协商、媒体通道建立后,再开始发送音频数据,避免数据丢失。
  • 生产环境优化:替换代码中的panic为合理的错误返回逻辑,避免程序崩溃。

内容的提问来源于stack exchange,提问作者shellfly

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.12 14:06:08