You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何基于OpenAI实时数据流生成实时文本转语音(TTS)

流式OpenAI响应同步生成ElevenLabs TTS的实现方案

我用OpenAI Chat API开发虚拟助手,API会分块流式返回响应来降低感知延迟。现在想在OpenAI生成内容的同时,用ElevenLabs Voice API做文本转语音,但目前代码要等内容全生成完才调用TTS,效果不好,想知道ElevenLabs或其他TTS服务是否支持这种场景,求技术方案。


服务器端代码

// Dependencies
const express = require("express");
const app = express();
const cors = require("cors");
const server = require("http").Server(app);
const { Configuration, OpenAIApi } = require("openai");
const { OpenAIExt } = require("openai-ext");
const voice = require('elevenlabs-node');
const fs = require('fs');

// Declare ejs, json formatting, set static files folder and initialise CORS.
app.set("view engine", "ejs");
app.set("json spaces", 2);
app.use(express.static("public"));
app.use(cors());

// Set the parser settings for JSON.
app.use(express.urlencoded({ extended: false }));
app.use(express.json());

// OpenAI Config
const configuration = new Configuration({
    apiKey: "",
});
const openai = new OpenAIApi(configuration);

// Set up Elevenlabs voice API
const apiKey = '';      // Your API key from Elevenlabs
const voiceID = 'pNInz6obpgDQGcFmaJgB';                 // The ID of the voice you want to get
const fileName = 'public/speech.mp3';                   // The name of your audio file

// Configure the stream (use type ServerStreamChatCompletionConfig for TypeScript users)
const streamConfig = {
    openai: openai,

    handler: {
        // Content contains the string draft, which may be partial. When isFinal is true, the completion is done.
        onContent(content, isFinal, stream) {
            console.log(content, "isFinal?", isFinal);
        },
        onDone(stream) {
          //  console.log("Done!");
            stream.destroy();
        },
        onError(error, stream) {    
            console.error(error);
        },
    },
};

// Set up SSE route for updates
app.get("/updates", (req, res) => {
    res.setHeader("Content-Type", "text/event-stream");
    res.setHeader("Cache-Control", "no-cache");
    res.setHeader("Connection", "keep-alive");
  
    // Send a comment to indicate that the connection was successful
    res.write(": connected\n\n");
  
    // Set up event listener for onContent updates
    streamConfig.handler.onContent = (content, isFinal, stream) => {
        try {
            const data = JSON.stringify({ content, isFinal });
      
            // Send the update to the client as an SSE event
            res.write(`event: update\ndata: ${data}\n\n`);

            if (isFinal == true) {
                voice.textToSpeech(apiKey, voiceID, fileName, content).then(res => {        // This is the closest I have been able to get,
                    console.log(res);                                                       // But this executes only once the content is done outputting
                });                                                                         // And still doesnt really work
            }


          } catch (error) {
            console.error("Error sending update:", error);
            res.status(500).end();
          }
    };
    streamConfig.handler.onDone = (stream) => {
        // console.log("Done!");        
        stream.destroy();
        res.end();
    }
    streamConfig.handler.onError = (error, stream) => {    
        console.error("Big bad error: " + error);
    };

  // Handle any errors that might occur while setting up the stream
  streamConfig.handler.onError = (error) => {
    console.error("Error setting up stream:", error);
    res.status(500).end();
  };
  });

app.post("/openai", async (req, res) => {
    const messages = req.body.messages;

    // Make the call to stream the completion
    const response = await OpenAIExt.streamServerChatCompletion(
        {
            model: "gpt-3.5-turbo",
            messages: messages,
            max_tokens: 1024,
            temperature: 1,
            top_p: 1,
            frequency_penalty: 0.0,
            presence_penalty: 0.6,
        },
        streamConfig 
    );

     // Send a success message back to the client
    res.json({ message: "Request successful" });
});

// Check the login status of the user, then display the index.html file in the home page.
app.get("/", (req, res) => {
    res.render("index");
});

app.get("/test", (req, res) => {
    res.render("index copy");
});

server.listen(3000);

客户端代码

// script.js

// References
const queryBox = document.getElementById("query-box");
const mainContent = document.getElementById("main-content");

// Speech Recognition
const SpeechRecognition = window.SpeechRecognition || window.webkitSpeechRecognition;
let recognition;
let recording = false;
let speaking = false;

const skylaKeywords = ["Skylar", "Skyler", "scholar"];

const messages = [
    { role: "system", content: "You are a virtual assistant named Skyla that is designed to speak like J.A.R.V.I.S from the Iron Man Movies, and respond as such by including a bit of his sass in responses. You can refer to me as Sir if you so wish." },
    { role: "user", content: "Skyla can you speak more like Jarvis" },
    { role: "assistant", content: "Of course, Sir. Is there a specific phrase or tone you would like me to emulate? Just let me know and I'll do my best to channel my inner J.A.R.V.I.S for you." },
    { role: "user", content: "I want you to speak like Jarvis from the first Iron Man movie, incorporating just a bit more of his Sass in responses" },
];

let questionNumber = 0;

// Start Recognition on page load.
addEventListener("load", (e) => {
    speechToText();
});

queryBox.addEventListener("keyup", function (e) {
    if (e.key === "Enter" && !e.shiftKey) {
        e.preventDefault();
        document.getElementById("submit-query").click();
    }
});

function submitQuery() {
    fetchResponse(queryBox.value);
    queryBox.style.height = "55px";
    queryBox.value = "";
}

const source = new EventSource("/updates");

source.addEventListener("open", () => {
    console.log("Connection to updates endpoint opened");
});

source.addEventListener("update", (event) => {
    const { content, isFinal } = JSON.parse(event.data);
    const queryBox = document.getElementById(questionNumber);

    speaking = true;
    mainContent.scrollTop = mainContent.scrollHeight;

    // Update the element with the new content
    if (queryBox != null) {
        queryBox.innerHTML = "<img src='icons/skyla.png'><div><p>" + content + "</p></div>";
    }

    if (isFinal) {
        console.log("Completion finished");

        const audio = new Audio(audioFile);
        audio.play();

        messages.push({ role: "assistant", content: content });
        questionNumber += 1;
        speaking = false;
    }
});

// Convert speech to text
function speechToText() {
    try {
        // Initialise Speech Recognition
        recognition = new SpeechRecognition();
        recognition.lang = "en";
        recognition.interimResults = true;

        // Start Recognition
        recognition.start();
        recognition.onresult = (event) => {
            let speech = event.results[0][0].transcript;

            // Replace 'Skylar, Skyler or Scholar' with Skyla
            skylaKeywords.forEach((keyword) => {
                if (speech === keyword && !recording) {
                    speech = speech.replaceAll(keyword, "Skyla");
                    queryBox.classList.add("recording");
                    recording = true;
                }
            });

            console.log(speech);

            // Detect the final speech result.
            if (event.results[0].isFinal && recording && speaking == false) {
                let newSpeech = speech;

                skylaKeywords.forEach((keyword) => {
                    if (speech.includes(keyword)) {
                        newSpeech = speech.replaceAll(keyword, "Skyla");
                    }
                });

                fetchResponse(newSpeech);
            }
        };
        recognition.onspeechend = () => {
            speechToText();
        };
        recognition.onerror = (event) => {
            stopRecording();

            switch (event.error) {
                case "no-speech":
                  speechToText();
                  break;
                case "audio-capture":
                  alert("No microphone was found. Ensure that a microphone is installed.");
                  break;
                case "not-allowed":
                  alert("Permission to use microphone is blocked.");
                  break;
                case "aborted":
                    alert("Listening Stopped.");
                    break;
                default:
                    alert("Error occurred in recognition: " + event.error);
                    break;
            }
        };
    } catch (error) {
        recording = false;

        console.log(error);
    }
}

function fetchResponse(content) {
    // Append the speech to the main-content div.
    const newInputElement = document.createElement("div");
    newInputElement.classList = "user-input content-box";
    newInputElement.innerHTML = "<img src='icons/avatar.png'><div><p>" + content + "</p></div>";
    mainContent.append(newInputElement);
    mainContent.scrollTop = mainContent.scrollHeight;

    messages.push({ role: "user", content: content });

    // fetch to the api.
    fetch("/openai", {
        method: "POST",
        headers: {
            "Content-Type": "application/json",
        },
        body: JSON.stringify({ messages: messages, }),
    }).then((response) => response.json())
      .then((data) => {

            // Append the speech to the main-content div.
            const newResponseElement = document.createElement("div");
            newResponseElement.classList = "skyla-response content-box";
            newResponseElement.id = questionNumber;
            newResponseElement.innerHTML = "<img src='icons/skyla.png'><p>" + data.data + "</p>";
            mainContent.append(newResponseElement);
        })
        .catch((error) => console.error(error));
}

// Stop Voice Recognition
function stopRecording() {
    queryBox.classList.remove("recording");
    recording = false;
}

解决方案

ElevenLabs支持流式TTS输出,不需要等完整文本生成,拿到片段就能请求生成音频流,再实时推给客户端播放,实现“边生成边朗读”的效果。

1. 服务器端改造:流式处理OpenAI响应+ElevenLabs流式TTS

替换第三方封装库,直接调用ElevenLabs的流式API,在OpenAI的流回调中按标点分割文本片段,实时请求TTS并推给客户端:

// 流式请求ElevenLabs TTS
async function streamTTS(textChunk) {
  const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceID}/stream`, {
    method: 'POST',
    headers: {
      'Content-Type': 'application/json',
      'xi-api-key': apiKey
    },
    body: JSON.stringify({
      text: textChunk,
      model_id: 'eleven_monolingual_v1',
      voice_settings: { stability: 0.5, similarity_boost: 0.5 }
    })
  });
  return response.body; // 返回可读音频流
}

// 改造OpenAI流回调
let textBuffer = '';
const punctuation = /[.!?。!?]/; // 按标点分割片段,保证语音流畅度

streamConfig.handler.onContent = (content, isFinal, stream) => {
  try {
    // 推送文本更新给客户端
    const data = JSON.stringify({ content, isFinal });
    res.write(`event: update\ndata: ${data}\n\n`);

    textBuffer += content;
    // 积累到完整句子或最后一段时,调用流式TTS
    if (punctuation.test(textBuffer) || isFinal) {
      streamTTS(textBuffer).then(audioStream => {
        const reader = audioStream.getReader();
        reader.read().then(function processChunk({ done, value }) {
          if (done) return;
          // 二进制音频转base64,通过SSE推给客户端
          const base64Audio = btoa(String.fromCharCode(...new Uint8Array(value)));
          res.write(`event: audio\ndata: ${base64Audio}\n\n`);
          return reader.read().then(processChunk);
        });
      });
      textBuffer = ''; // 清空缓存
    }

    if (isFinal) res.end();
  } catch (error) {
    console.error("Error sending update:", error);
    res.status(500).end();
  }
};

2. 客户端改造:实时接收并播放音频流

监听新的audio事件,用AudioContext解码并播放音频片段:

const source = new EventSource("/updates");
const audioContext = new (window.AudioContext || window.webkitAudioContext)();
let audioBufferQueue = [];
let isPlaying = false;

// 处理流式音频
source.addEventListener("audio", async (event) => {
  const base64Audio = event.data;
  const audioData = Uint8Array.from(atob(base64Audio), c => c.charCodeAt(0));
  
  // 解码音频数据
  const audioBuffer = await audioContext.decodeAudioData(audioData.buffer);
  audioBufferQueue.push(audioBuffer);
  
  // 队列非空且未播放时,启动播放
  if (!isPlaying) playAudioQueue();
});

// 播放音频队列
async function playAudioQueue() {
  isPlaying = true;
  while (audioBufferQueue.length > 0) {
    const buffer = audioBufferQueue.shift();
    const sourceNode = audioContext.createBufferSource();
    sourceNode.buffer = buffer;
    sourceNode.connect(audioContext.destination);
    await new Promise(resolve => {
      sourceNode.onended = resolve;
      sourceNode.start();
    });
  }
  isPlaying = false;
}

// 原文本更新逻辑保持不变
source.addEventListener("update", (event) => {
  const { content, isFinal } = JSON.parse(event.data);
  const queryBox = document.getElementById(questionNumber);

  speaking = true;
  mainContent.scrollTop = mainContent.scrollHeight;

  if (queryBox != null) {
    queryBox.innerHTML = "<img src='icons/skyla.png'><div><p>" + content + "</p></div>";
  }

  if (isFinal) {
    console.log("Completion finished");
    messages.push({ role: "assistant", content: content });
    questionNumber += 1;
    speaking = false;
  }
});

3. 注意事项

  • 文本片段分割:不要用单个词的细碎片段,按句子或标点分割,平衡实时性和语音流畅度。
  • API限流:ElevenLabs流式API有请求频率限制,控制片段发送频率避免触发限流。
  • 音频播放优化:可以用MediaSource API替代队列播放,实现更无缝的流式音频体验。
  • 错误处理:添加OpenAI流中断、ElevenLabs请求失败的捕获逻辑,保证客户端稳定性。

内容的提问来源于stack exchange,提问作者Skyfall106

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.23 23:33:07