如何基于Web Speech API实现文本转语音音频下载功能?
如何为Web Speech API实现的文本转语音工具添加音频下载功能?
我用JavaScript的Web Speech API实现了文本转语音功能,现在想让用户能下载转换生成的音频,以下是我的现有代码:
const textarea = document.querySelector("textarea"), voiceList = document.querySelector("select"), speechBtn = document.querySelector("button"); let synth = speechSynthesis, isSpeaking = true; voices(); function voices(){ for(let voice of synth.getVoices()){ let selected = voice.name === "Google US English" ? "selected" : ""; let option = `<option value="${voice.name}" ${selected}>${voice.name} (${voice.lang})</option>`; voiceList.insertAdjacentHTML("beforeend", option); } } synth.addEventListener("voiceschanged", voices); function textToSpeech(text){ let utterance = new SpeechSynthesisUtterance(text); for(let voice of synth.getVoices()){ if(voice.name === voiceList.value){ utterance.voice = voice; } } synth.speak(utterance); } speechBtn.addEventListener("click", e =>{ e.preventDefault(); if(textarea.value !== ""){ if(!synth.speaking){ textToSpeech(textarea.value); } if(textarea.value.length > 80){ setInterval(()=>{ if(!synth.speaking && !isSpeaking){ isSpeaking = true; speechBtn.innerText = "Convert To Speech"; }else{ } }, 500); if(isSpeaking){ synth.resume(); isSpeaking = false; speechBtn.innerText = "Pause Speech"; }else{ synth.pause(); isSpeaking = true; speechBtn.innerText = "Resume Speech"; } }else{ speechBtn.innerText = "Convert To Speech"; } } });
请问是否可以为这个工具添加音频下载功能?
可以实现,但Web Speech API本身不直接提供导出音频的接口——它是在浏览器本地合成语音,没有暴露原始音频数据。我们可以通过MediaRecorder API录制语音合成的输出,再将录制内容转换成可下载的音频文件。以下是具体修改方案:
1. 先在HTML中添加下载按钮
在现有按钮旁新增一个下载按钮:
<button id="downloadBtn" disabled>下载音频</button>
2. 修改JavaScript代码,实现录制与下载功能
const textarea = document.querySelector("textarea"), voiceList = document.querySelector("select"), speechBtn = document.querySelector("button"), downloadBtn = document.getElementById("downloadBtn"); let synth = speechSynthesis, isSpeaking = true, mediaRecorder = null, audioChunks = []; voices(); function voices(){ for(let voice of synth.getVoices()){ let selected = voice.name === "Google US English" ? "selected" : ""; let option = `<option value="${voice.name}" ${selected}>${voice.name} (${voice.lang})</option>`; voiceList.insertAdjacentHTML("beforeend", option); } } synth.addEventListener("voiceschanged", voices); function textToSpeech(text){ // 创建音频上下文和媒体流目标,捕获语音合成输出 const audioContext = new (window.AudioContext || window.webkitAudioContext)(); const destination = audioContext.createMediaStreamDestination(); const mediaStream = destination.stream; // 初始化MediaRecorder并开始录制 mediaRecorder = new MediaRecorder(mediaStream); audioChunks = []; mediaRecorder.ondataavailable = (e) => { if (e.data.size > 0) audioChunks.push(e.data); }; mediaRecorder.onstop = () => { // 录制结束后生成音频Blob并触发下载 const audioBlob = new Blob(audioChunks, { type: 'audio/webm' }); const audioUrl = URL.createObjectURL(audioBlob); const a = document.createElement('a'); a.href = audioUrl; a.download = 'text-to-speech.webm'; a.click(); URL.revokeObjectURL(audioUrl); // 释放临时资源 downloadBtn.disabled = true; }; mediaRecorder.start(); // 将语音合成输出导向媒体流目标 const utterance = new SpeechSynthesisUtterance(text); const utteranceStream = audioContext.createMediaStreamSource(mediaStream); utteranceStream.connect(destination); for(let voice of synth.getVoices()){ if(voice.name === voiceList.value){ utterance.voice = voice; } } // 语音合成结束时停止录制 utterance.onend = () => { setTimeout(() => { if(mediaRecorder && mediaRecorder.state !== 'inactive'){ mediaRecorder.stop(); } }, 100); // 延迟确保所有音频片段被捕获 }; synth.speak(utterance); downloadBtn.disabled = false; } speechBtn.addEventListener("click", e =>{ e.preventDefault(); if(textarea.value !== ""){ if(!synth.speaking){ textToSpeech(textarea.value); } if(textarea.value.length > 80){ setInterval(()=>{ if(!synth.speaking && !isSpeaking){ isSpeaking = true; speechBtn.innerText = "Convert To Speech"; // 语音结束后停止录制 if(mediaRecorder && mediaRecorder.state !== 'inactive'){ mediaRecorder.stop(); } }else{ } }, 500); if(isSpeaking){ synth.resume(); isSpeaking = false; speechBtn.innerText = "Pause Speech"; // 同步恢复录制 if(mediaRecorder && mediaRecorder.state === 'paused'){ mediaRecorder.resume(); } }else{ synth.pause(); isSpeaking = true; speechBtn.innerText = "Resume Speech"; // 同步暂停录制 if(mediaRecorder && mediaRecorder.state === 'recording'){ mediaRecorder.pause(); } } }else{ speechBtn.innerText = "Convert To Speech"; } } }); // 手动触发下载(可选,也可在语音结束自动触发) downloadBtn.addEventListener('click', () => { if(mediaRecorder && mediaRecorder.state === 'recording'){ mediaRecorder.stop(); } });
关键说明
- 音频捕获逻辑:通过
AudioContext将语音合成的输出导向MediaStreamDestination,再用MediaRecorder录制这个流,实现对合成语音的捕获。 - 状态同步:修改原有的暂停/恢复逻辑,同步控制录制状态,保证音频片段的完整性。
- 下载处理:录制结束后将音频片段合并成Blob,生成临时URL,通过隐藏的a标签触发浏览器下载。
注意事项
- 大部分浏览器要求MediaRecorder必须在用户交互(如点击事件)内初始化,所以我们在
textToSpeech函数中创建实例,符合浏览器安全要求。 - 生成的音频格式为webm,现代浏览器普遍支持;如果需要mp3等格式,需额外引入编码库转换,但webm是最省心的原生选择。
- 长文本场景下,建议测试暂停/恢复功能是否能同步录制状态,避免音频缺失。
内容的提问来源于stack exchange,提问作者abigail daniel
相关产品推荐
相关产品推荐

