如何将Azure TTS生成的OGG数据写入WebRTC音频轨道?代码排查
问题描述
使用Azure Speech SDK实现文本转语音(TTS)生成OGG格式数据,流式输出发送至WebRTC音频轨道,数据能成功发送到对等端,但客户端听不到音频。相关代码如下:
发送音频数据的函数
func sendAudioData(data io.Reader, audioTrack *webrtc.TrackLocalStaticSample) error { ogg, _, oggErr := oggreader.NewWith(data) if oggErr != nil { return oggErr } var lastGranule uint64 for { pageData, pageHeader, oggErr := ogg.ParseNextPage() if errors.Is(oggErr, io.EOF) { fmt.Printf("All audio pages parsed and sent") return nil } if oggErr != nil { return oggErr } slog.Info("ParseNextPage", "header", pageHeader) // The amount of samples is the difference between the last and current timestamp sampleCount := float64(pageHeader.GranulePosition - lastGranule) lastGranule = pageHeader.GranulePosition sampleDuration := time.Duration((sampleCount/48000)*1000) * time.Millisecond slog.Info("Write Sample", "duration", sampleDuration, "length", len(pageData)) if oggErr = audioTrack.WriteSample(media.Sample{Data: pageData, Duration: sampleDuration}); oggErr != nil { return oggErr } } }
生成流式数据的代码
package speech import ( "errors" "fmt" "os" "time" "github.com/Microsoft/cognitive-services-speech-sdk-go/audio" "github.com/Microsoft/cognitive-services-speech-sdk-go/common" "github.com/Microsoft/cognitive-services-speech-sdk-go/speech" ) func azureTTSStream(text, lang string) (*speech.AudioDataStream, error) { // stream, err := audio.CreatePullAudioOutputStream() // if err != nil { // return nil, err // } // audioConfig, err := audio.NewAudioConfigFromStreamOutput(stream) // if err != nil { // panic(err) // } speechKey := os.Getenv("AZURE_SPEECH_KEY") speechRegion := os.Getenv("AZURE_SPEECH_REGION") speechConfig, err := speech.NewSpeechConfigFromSubscription(speechKey, speechRegion) if err != nil { panic(err) } speechConfig.SetSpeechSynthesisOutputFormat(common.Ogg48Khz16BitMonoOpus) synthesizer, err := speech.NewSpeechSynthesizerFromConfig(speechConfig, nil) if err != nil { panic(err) } voice := azureVoices[lang] input := fmt.Sprintf(azureSAMLInput, lang, voice, text) task := synthesizer.SpeakSsmlAsync(input) var outcome speech.SpeechSynthesisOutcome select { case outcome = <-task: case <-time.After(60 * time.Second): fmt.Println("timeout") return nil, errors.New("timeout") } defer outcome.Close() if outcome.Error != nil { fmt.Println("outcome error: ", outcome.Error) } return speech.NewAudioDataStreamFromSpeechSynthesisResult(outcome.Result) // if outcome.Result.Reason == common.SynthesizingAudioCompleted { // fmt.Println("outcome completed") // } // cancellation, _ := speech.NewCancellationDetailsFromSpeechSynthesisResult(outcome.Result) // msg := fmt.Sprintf("CANCELED: Reason=%d.\n", cancellation.Reason) // if cancellation.Reason == common.Error { // fmt.Sprintf("%s ErrorCode=%d, ErrorDetails=[%s]\n", msg, cancellation.ErrorCode, cancellation.ErrorDetails) // } }
问题排查与修复方案
核心问题分析
- OGG容器数据直接写入错误:WebRTC音频轨道需要裸Opus编码帧,但你直接将包含OGG容器头的页面数据
pageData写入轨道,客户端无法解析容器格式,自然听不到声音。 - 时长计算逻辑错误:Azure TTS输出的OGG Opus中,GranulePosition并非直接对应PCM样本数,用固定48000采样率计算时长会导致时间轴混乱,引发播放不同步或静音。
- 错误处理与资源管理缺失:原代码未检查TTS合成是否成功完成,也未正确关闭合成器资源,可能导致数据流不完整或格式损坏。
修复后的代码
1. 修正sendAudioData函数(提取裸Opus帧)
使用opusreader解析OGG页面中的纯Opus帧,再写入WebRTC轨道:
import ( "io" "errors" "time" "github.com/pion/webrtc/v3" "github.com/pion/webrtc/v3/media" "github.com/pion/opus" "github.com/pion/oggreader" ) func sendAudioData(data io.Reader, audioTrack *webrtc.TrackLocalStaticSample) error { ogg, _, err := oggreader.NewWith(data) if err != nil { return err } // 初始化Opus解码器,用于计算单帧时长 opusDecoder, err := opus.NewDecoder(48000, 1) if err != nil { return err } for { pageData, pageHeader, err := ogg.ParseNextPage() if errors.Is(err, io.EOF) { return nil } if err != nil { return err } // 从OGG页面中提取所有裸Opus帧 opusFrames, err := opusreader.ParseFrames(pageData) if err != nil { return err } for _, frame := range opusFrames { // 计算当前Opus帧的精确时长 frameSamples, err := opusDecoder.FrameSamples(frame) if err != nil { return err } duration := time.Duration((float64(frameSamples)/48000)*1000) * time.Millisecond // 将裸Opus帧写入WebRTC轨道 if err := audioTrack.WriteSample(media.Sample{ Data: frame, Duration: duration, }); err != nil { return err } } } }
2. 修正azureTTSStream函数(完善错误处理与资源管理)
补充必要的参数校验、合成结果检查,确保数据流完整:
package speech import ( "errors" "fmt" "os" "time" "github.com/Microsoft/cognitive-services-speech-sdk-go/common" "github.com/Microsoft/cognitive-services-speech-sdk-go/speech" ) func azureTTSStream(text, lang string) (*speech.AudioDataStream, error) { // 检查环境变量是否配置 speechKey := os.Getenv("AZURE_SPEECH_KEY") speechRegion := os.Getenv("AZURE_SPEECH_REGION") if speechKey == "" || speechRegion == "" { return nil, errors.New("AZURE_SPEECH_KEY或AZURE_SPEECH_REGION环境变量未设置") } speechConfig, err := speech.NewSpeechConfigFromSubscription(speechKey, speechRegion) if err != nil { return nil, err } // 明确指定输出格式为OGG Opus 48kHz单声道 speechConfig.SetSpeechSynthesisOutputFormat(common.Ogg48Khz16BitMonoOpus) synthesizer, err := speech.NewSpeechSynthesizerFromConfig(speechConfig, nil) if err != nil { return nil, err } defer synthesizer.Close() // 确保合成器资源被正确释放 // 检查语言对应的语音是否存在 voice, ok := azureVoices[lang] if !ok { return nil, fmt.Errorf("不支持的语言:%s", lang) } input := fmt.Sprintf(azureSAMLInput, lang, voice, text) task := synthesizer.SpeakSsmlAsync(input) var outcome speech.SpeechSynthesisOutcome select { case outcome = <-task: case <-time.After(60 * time.Second): return nil, errors.New("TTS合成超时") } defer outcome.Close() // 检查合成是否出错 if outcome.Error != nil { return nil, fmt.Errorf("TTS合成失败:%w", outcome.Error) } // 确认合成已完成 if outcome.Result.Reason != common.SynthesizingAudioCompleted { return nil, errors.New("TTS合成未正常完成") } return speech.NewAudioDataStreamFromSpeechSynthesisResult(outcome.Result), nil }
额外注意事项
- WebRTC轨道配置:创建音频轨道时必须指定Opus编解码器,且参数与Azure TTS输出匹配:
audioTrack, err := webrtc.NewTrackLocalStaticSample(webrtc.RTPCodecCapability{MimeType: webrtc.MimeTypeOpus}, "audio", "pion") - 传输时机:确保WebRTC PeerConnection完成ICE协商、媒体通道建立后,再开始发送音频数据,避免数据丢失。
- 生产环境优化:替换代码中的
panic为合理的错误返回逻辑,避免程序崩溃。
内容的提问来源于stack exchange,提问作者shellfly
相关产品推荐
相关产品推荐

