You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Go中基于字符平均音频长度截断coqui-ai TTS的过长音频?

解决方案

要实现基于平均字符音频长度截断异常短句音频的功能,你需要按以下步骤修改代码:

  1. 获取音频时长:通过soxi命令提取每个音频文件的时长(需确保系统已安装sox工具)。
  2. 计算平均字符时长:统计所有有效音频的总字符数和总时长,算出单个字符对应的平均音频时长。
  3. 截断异常音频:对比单句音频的实际时长与预期时长(字符数×平均字符时长),若超出设定阈值则截断到预期时长。
  4. 合并处理后的音频:使用处理后的音频文件进行合并。

以下是修改后的完整代码:

package main

import (
	"log"
	"os"
	"os/exec"
	"path/filepath"
	"strconv"
	"strings"

	"github.com/cheggaaa/pb/v3"
	"github.com/neurosnap/sentences/english"
)

type audioInfo struct {
	filePath  string
	charCount int
}

func main() {
	if len(os.Args) != 2 {
		log.Fatalf("Usage: go run main.go <input>")
	}
	sentences := getSentences()
	audioInfos := convertTextToAudio(sentences)
	processedFiles := processAbnormalAudio(audioInfos)
	concatenateAudioFiles(processedFiles)
	cleanupTempFiles(processedFiles)
}

func getSentences() []string {
	tokenizer, err := english.NewSentenceTokenizer(nil)
	if err != nil {
		panic(err)
	}
	text, err := os.ReadFile(os.Args[1])
	if err != nil {
		log.Fatal(err)
	}
	tmp := tokenizer.Tokenize(string(text))
	var sentences []string
	for _, sentence := range tmp {
		sentences = append(sentences, sentence.Text)
	}
	return sentences
}

func convertTextToAudio(sentences []string) []audioInfo {
	var audioInfos []audioInfo
	bar := pb.StartNew(len(sentences))
	for i, sentence := range sentences {
		audioFile := "out_" + strconv.Itoa(i) + ".wav"
		cmd := exec.Command("tts", "--text", sentence, "--model_name", "tts_models/en/ljspeech/tacotron2-DDC", "--out_path", audioFile)
		err := cmd.Run()
		if err != nil {
			log.Println(cmd.String())
			log.Printf("Failed to convert sentence: %s", sentence)
		} else {
			audioInfos = append(audioInfos, audioInfo{
				filePath:  audioFile,
				charCount: len(sentence),
			})
		}
		bar.Increment()
	}
	bar.Finish()
	return audioInfos
}

func getAudioDuration(filePath string) (float64, error) {
	cmd := exec.Command("soxi", "-D", filePath)
	output, err := cmd.Output()
	if err != nil {
		return 0, err
	}
	return strconv.ParseFloat(strings.TrimSpace(string(output)), 64)
}

func processAbnormalAudio(audioInfos []audioInfo) []string {
	if len(audioInfos) == 0 {
		log.Fatal("No valid audio files to process")
	}

	// 计算总字符数和总时长
	var totalChars int
	var totalDuration float64
	for _, info := range audioInfos {
		dur, err := getAudioDuration(info.filePath)
		if err != nil {
			log.Printf("Failed to get duration for %s: %v", info.filePath, err)
			continue
		}
		totalChars += info.charCount
		totalDuration += dur
	}

	if totalChars == 0 {
		log.Fatal("No valid audio data to calculate average duration")
	}
	avgCharDuration := totalDuration / float64(totalChars)
	threshold := 1.5 // 允许实际时长超出预期的最大倍数(可调整)

	var processedFiles []string
	for _, info := range audioInfos {
		dur, err := getAudioDuration(info.filePath)
		if err != nil {
			log.Printf("Skipping %s: %v", info.filePath, err)
			continue
		}

		expectedDur := float64(info.charCount) * avgCharDuration
		if dur > expectedDur*threshold {
			// 生成截断后的文件名
			truncatedFile := "truncated_" + filepath.Base(info.filePath)
			// 使用sox截断音频到预期时长
			cmd := exec.Command("sox", info.filePath, truncatedFile, "trim", "0", strconv.FormatFloat(expectedDur, 'f', 2, 64))
			if err := cmd.Run(); err != nil {
				log.Printf("Failed to truncate %s: %v", info.filePath, err)
				processedFiles = append(processedFiles, info.filePath)
			} else {
				processedFiles = append(processedFiles, truncatedFile)
				// 删除原异常音频文件
				os.Remove(info.filePath)
			}
		} else {
			processedFiles = append(processedFiles, info.filePath)
		}
	}

	return processedFiles
}

func concatenateAudioFiles(audioFiles []string) {
	if len(audioFiles) == 0 {
		log.Fatal("No files to concatenate")
	}
	// 拼接输出文件到参数列表
	cmdArgs := append(audioFiles, "out.wav")
	cmd := exec.Command("sox", cmdArgs...)
	if err := cmd.Run(); err != nil {
		log.Fatalf("Failed to run sox: %v", err)
	}
}

func cleanupTempFiles(processedFiles []string) {
	for _, file := range processedFiles {
		if strings.HasPrefix(filepath.Base(file), "truncated_") || strings.HasPrefix(filepath.Base(file), "out_") {
			os.Remove(file)
		}
	}
}

关键说明

  • audioInfo结构体:存储音频文件路径和对应句子的字符数,方便后续关联计算。
  • getAudioDuration:调用soxi -D命令获取音频时长,返回浮点型秒数。
  • processAbnormalAudio:核心处理逻辑,先计算平均字符时长,再对比实际与预期时长,超出阈值则用sox trim命令截断。
  • cleanupTempFiles:可选的临时文件清理函数,删除转换和截断过程中生成的中间文件。

你可以根据实际情况调整threshold参数(比如设为1.2表示允许20%的时长偏差),确保截断后的音频更自然。

内容的提问来源于stack exchange,提问作者user20317520

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.15 10:15:37