如何基于OpenAI实时数据流生成实时文本转语音(TTS)
流式OpenAI响应同步生成ElevenLabs TTS的实现方案
我用OpenAI Chat API开发虚拟助手,API会分块流式返回响应来降低感知延迟。现在想在OpenAI生成内容的同时,用ElevenLabs Voice API做文本转语音,但目前代码要等内容全生成完才调用TTS,效果不好,想知道ElevenLabs或其他TTS服务是否支持这种场景,求技术方案。
服务器端代码
// Dependencies const express = require("express"); const app = express(); const cors = require("cors"); const server = require("http").Server(app); const { Configuration, OpenAIApi } = require("openai"); const { OpenAIExt } = require("openai-ext"); const voice = require('elevenlabs-node'); const fs = require('fs'); // Declare ejs, json formatting, set static files folder and initialise CORS. app.set("view engine", "ejs"); app.set("json spaces", 2); app.use(express.static("public")); app.use(cors()); // Set the parser settings for JSON. app.use(express.urlencoded({ extended: false })); app.use(express.json()); // OpenAI Config const configuration = new Configuration({ apiKey: "", }); const openai = new OpenAIApi(configuration); // Set up Elevenlabs voice API const apiKey = ''; // Your API key from Elevenlabs const voiceID = 'pNInz6obpgDQGcFmaJgB'; // The ID of the voice you want to get const fileName = 'public/speech.mp3'; // The name of your audio file // Configure the stream (use type ServerStreamChatCompletionConfig for TypeScript users) const streamConfig = { openai: openai, handler: { // Content contains the string draft, which may be partial. When isFinal is true, the completion is done. onContent(content, isFinal, stream) { console.log(content, "isFinal?", isFinal); }, onDone(stream) { // console.log("Done!"); stream.destroy(); }, onError(error, stream) { console.error(error); }, }, }; // Set up SSE route for updates app.get("/updates", (req, res) => { res.setHeader("Content-Type", "text/event-stream"); res.setHeader("Cache-Control", "no-cache"); res.setHeader("Connection", "keep-alive"); // Send a comment to indicate that the connection was successful res.write(": connected\n\n"); // Set up event listener for onContent updates streamConfig.handler.onContent = (content, isFinal, stream) => { try { const data = JSON.stringify({ content, isFinal }); // Send the update to the client as an SSE event res.write(`event: update\ndata: ${data}\n\n`); if (isFinal == true) { voice.textToSpeech(apiKey, voiceID, fileName, content).then(res => { // This is the closest I have been able to get, console.log(res); // But this executes only once the content is done outputting }); // And still doesnt really work } } catch (error) { console.error("Error sending update:", error); res.status(500).end(); } }; streamConfig.handler.onDone = (stream) => { // console.log("Done!"); stream.destroy(); res.end(); } streamConfig.handler.onError = (error, stream) => { console.error("Big bad error: " + error); }; // Handle any errors that might occur while setting up the stream streamConfig.handler.onError = (error) => { console.error("Error setting up stream:", error); res.status(500).end(); }; }); app.post("/openai", async (req, res) => { const messages = req.body.messages; // Make the call to stream the completion const response = await OpenAIExt.streamServerChatCompletion( { model: "gpt-3.5-turbo", messages: messages, max_tokens: 1024, temperature: 1, top_p: 1, frequency_penalty: 0.0, presence_penalty: 0.6, }, streamConfig ); // Send a success message back to the client res.json({ message: "Request successful" }); }); // Check the login status of the user, then display the index.html file in the home page. app.get("/", (req, res) => { res.render("index"); }); app.get("/test", (req, res) => { res.render("index copy"); }); server.listen(3000);
客户端代码
// script.js // References const queryBox = document.getElementById("query-box"); const mainContent = document.getElementById("main-content"); // Speech Recognition const SpeechRecognition = window.SpeechRecognition || window.webkitSpeechRecognition; let recognition; let recording = false; let speaking = false; const skylaKeywords = ["Skylar", "Skyler", "scholar"]; const messages = [ { role: "system", content: "You are a virtual assistant named Skyla that is designed to speak like J.A.R.V.I.S from the Iron Man Movies, and respond as such by including a bit of his sass in responses. You can refer to me as Sir if you so wish." }, { role: "user", content: "Skyla can you speak more like Jarvis" }, { role: "assistant", content: "Of course, Sir. Is there a specific phrase or tone you would like me to emulate? Just let me know and I'll do my best to channel my inner J.A.R.V.I.S for you." }, { role: "user", content: "I want you to speak like Jarvis from the first Iron Man movie, incorporating just a bit more of his Sass in responses" }, ]; let questionNumber = 0; // Start Recognition on page load. addEventListener("load", (e) => { speechToText(); }); queryBox.addEventListener("keyup", function (e) { if (e.key === "Enter" && !e.shiftKey) { e.preventDefault(); document.getElementById("submit-query").click(); } }); function submitQuery() { fetchResponse(queryBox.value); queryBox.style.height = "55px"; queryBox.value = ""; } const source = new EventSource("/updates"); source.addEventListener("open", () => { console.log("Connection to updates endpoint opened"); }); source.addEventListener("update", (event) => { const { content, isFinal } = JSON.parse(event.data); const queryBox = document.getElementById(questionNumber); speaking = true; mainContent.scrollTop = mainContent.scrollHeight; // Update the element with the new content if (queryBox != null) { queryBox.innerHTML = "<img src='icons/skyla.png'><div><p>" + content + "</p></div>"; } if (isFinal) { console.log("Completion finished"); const audio = new Audio(audioFile); audio.play(); messages.push({ role: "assistant", content: content }); questionNumber += 1; speaking = false; } }); // Convert speech to text function speechToText() { try { // Initialise Speech Recognition recognition = new SpeechRecognition(); recognition.lang = "en"; recognition.interimResults = true; // Start Recognition recognition.start(); recognition.onresult = (event) => { let speech = event.results[0][0].transcript; // Replace 'Skylar, Skyler or Scholar' with Skyla skylaKeywords.forEach((keyword) => { if (speech === keyword && !recording) { speech = speech.replaceAll(keyword, "Skyla"); queryBox.classList.add("recording"); recording = true; } }); console.log(speech); // Detect the final speech result. if (event.results[0].isFinal && recording && speaking == false) { let newSpeech = speech; skylaKeywords.forEach((keyword) => { if (speech.includes(keyword)) { newSpeech = speech.replaceAll(keyword, "Skyla"); } }); fetchResponse(newSpeech); } }; recognition.onspeechend = () => { speechToText(); }; recognition.onerror = (event) => { stopRecording(); switch (event.error) { case "no-speech": speechToText(); break; case "audio-capture": alert("No microphone was found. Ensure that a microphone is installed."); break; case "not-allowed": alert("Permission to use microphone is blocked."); break; case "aborted": alert("Listening Stopped."); break; default: alert("Error occurred in recognition: " + event.error); break; } }; } catch (error) { recording = false; console.log(error); } } function fetchResponse(content) { // Append the speech to the main-content div. const newInputElement = document.createElement("div"); newInputElement.classList = "user-input content-box"; newInputElement.innerHTML = "<img src='icons/avatar.png'><div><p>" + content + "</p></div>"; mainContent.append(newInputElement); mainContent.scrollTop = mainContent.scrollHeight; messages.push({ role: "user", content: content }); // fetch to the api. fetch("/openai", { method: "POST", headers: { "Content-Type": "application/json", }, body: JSON.stringify({ messages: messages, }), }).then((response) => response.json()) .then((data) => { // Append the speech to the main-content div. const newResponseElement = document.createElement("div"); newResponseElement.classList = "skyla-response content-box"; newResponseElement.id = questionNumber; newResponseElement.innerHTML = "<img src='icons/skyla.png'><p>" + data.data + "</p>"; mainContent.append(newResponseElement); }) .catch((error) => console.error(error)); } // Stop Voice Recognition function stopRecording() { queryBox.classList.remove("recording"); recording = false; }
解决方案
ElevenLabs支持流式TTS输出,不需要等完整文本生成,拿到片段就能请求生成音频流,再实时推给客户端播放,实现“边生成边朗读”的效果。
1. 服务器端改造:流式处理OpenAI响应+ElevenLabs流式TTS
替换第三方封装库,直接调用ElevenLabs的流式API,在OpenAI的流回调中按标点分割文本片段,实时请求TTS并推给客户端:
// 流式请求ElevenLabs TTS async function streamTTS(textChunk) { const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceID}/stream`, { method: 'POST', headers: { 'Content-Type': 'application/json', 'xi-api-key': apiKey }, body: JSON.stringify({ text: textChunk, model_id: 'eleven_monolingual_v1', voice_settings: { stability: 0.5, similarity_boost: 0.5 } }) }); return response.body; // 返回可读音频流 } // 改造OpenAI流回调 let textBuffer = ''; const punctuation = /[.!?。!?]/; // 按标点分割片段,保证语音流畅度 streamConfig.handler.onContent = (content, isFinal, stream) => { try { // 推送文本更新给客户端 const data = JSON.stringify({ content, isFinal }); res.write(`event: update\ndata: ${data}\n\n`); textBuffer += content; // 积累到完整句子或最后一段时,调用流式TTS if (punctuation.test(textBuffer) || isFinal) { streamTTS(textBuffer).then(audioStream => { const reader = audioStream.getReader(); reader.read().then(function processChunk({ done, value }) { if (done) return; // 二进制音频转base64,通过SSE推给客户端 const base64Audio = btoa(String.fromCharCode(...new Uint8Array(value))); res.write(`event: audio\ndata: ${base64Audio}\n\n`); return reader.read().then(processChunk); }); }); textBuffer = ''; // 清空缓存 } if (isFinal) res.end(); } catch (error) { console.error("Error sending update:", error); res.status(500).end(); } };
2. 客户端改造:实时接收并播放音频流
监听新的audio事件,用AudioContext解码并播放音频片段:
const source = new EventSource("/updates"); const audioContext = new (window.AudioContext || window.webkitAudioContext)(); let audioBufferQueue = []; let isPlaying = false; // 处理流式音频 source.addEventListener("audio", async (event) => { const base64Audio = event.data; const audioData = Uint8Array.from(atob(base64Audio), c => c.charCodeAt(0)); // 解码音频数据 const audioBuffer = await audioContext.decodeAudioData(audioData.buffer); audioBufferQueue.push(audioBuffer); // 队列非空且未播放时,启动播放 if (!isPlaying) playAudioQueue(); }); // 播放音频队列 async function playAudioQueue() { isPlaying = true; while (audioBufferQueue.length > 0) { const buffer = audioBufferQueue.shift(); const sourceNode = audioContext.createBufferSource(); sourceNode.buffer = buffer; sourceNode.connect(audioContext.destination); await new Promise(resolve => { sourceNode.onended = resolve; sourceNode.start(); }); } isPlaying = false; } // 原文本更新逻辑保持不变 source.addEventListener("update", (event) => { const { content, isFinal } = JSON.parse(event.data); const queryBox = document.getElementById(questionNumber); speaking = true; mainContent.scrollTop = mainContent.scrollHeight; if (queryBox != null) { queryBox.innerHTML = "<img src='icons/skyla.png'><div><p>" + content + "</p></div>"; } if (isFinal) { console.log("Completion finished"); messages.push({ role: "assistant", content: content }); questionNumber += 1; speaking = false; } });
3. 注意事项
- 文本片段分割:不要用单个词的细碎片段,按句子或标点分割,平衡实时性和语音流畅度。
- API限流:ElevenLabs流式API有请求频率限制,控制片段发送频率避免触发限流。
- 音频播放优化:可以用MediaSource API替代队列播放,实现更无缝的流式音频体验。
- 错误处理:添加OpenAI流中断、ElevenLabs请求失败的捕获逻辑,保证客户端稳定性。
内容的提问来源于stack exchange,提问作者Skyfall106
相关产品推荐
相关产品推荐

