NodeJS通过WebRTC转发GPT-4o-audio-preview音频流异常排查
问题:NodeJS通过WebRTC转发GPT-4o-audio-preview音频流播放不稳定
我在NodeJS应用里尝试把OpenAI的gpt-4o-audio-preview模型生成的音频通过WebRTC(使用flutter_webrtc)转发给客户端,为此编写了AudioStreamer类,负责转换模型返回的音频delta、将采样率从24kHz上采样至48kHz,并把缓冲区拆分为接近实时的流。但音频播放效果不稳定:有时能完整正常播放,有时仅能听到开头几个词,随后迅速变为机械音并停止。
handleLLMAudio方法直接接收OpenAI库的输出,传入的是event.choices[0]?.delta?.audio?.data的值。
我的AudioStreamer类:
const wrtc = require("@roamhq/wrtc"); class AudioStreamer { constructor() { this.pc = null; this.inputSampleRate = 24000; // gpt-4o-audio-preview输出采样率 this.outputSampleRate = 48000; // WebRTC采样率 this.samplesPerFrame = 480; // PCM16格式下,960字节=480个采样点 this.audioBuffer = new Float32Array(0); this.track = null; this.mediaStream = null; this.isPlaying = false; this.lastPlayTime = 0; // 计算每帧的持续时间(毫秒) this.frameDuration = (this.samplesPerFrame / this.outputSampleRate) * 1000; // ~10ms } /** * 新的RTCPeerConnection和RTCAudioSource创建后调用 * 调用此方法前会先执行this.setPeerConnection * @param {RTCAudioSource} audioSource */ async initialize(audioSource) { const { MediaStream } = wrtc; this.audioSource = audioSource; this.track = this.audioSource.createTrack(); this.pc.addTrack(this.track); this.mediaStream = new MediaStream([this.audioSource]); } /** * 接收base64格式的音频字符串,开始转换为实时音频流 * @param {String} base64Audio 直接来自gpt-4o-audio-preview的PCM16音频delta */ handleLLMAudio(base64Audio) { const buffer = Buffer.from(base64Audio, 'base64'); const view = new DataView(buffer.buffer, buffer.byteOffset, buffer.length); const pcm16 = new Float32Array(buffer.length / 2); for (let i = 0; i < pcm16.length; i++) { const int16Value = view.getInt16(i * 2, true); pcm16[i] = int16Value / 32768.0; } const resampled = this.resampleBuffer(pcm16); // 添加到缓冲区 const newBuffer = new Float32Array(this.audioBuffer.length + resampled.length); newBuffer.set(this.audioBuffer); newBuffer.set(resampled, this.audioBuffer.length); this.audioBuffer = newBuffer; // 如果未开始播放则启动 if (!this.isPlaying) { this.isPlaying = true; this.lastPlayTime = Date.now(); this.processAudioBuffer(); } } resampleBuffer(inputBuffer) { const outputLength = Math.ceil(inputBuffer.length * (this.outputSampleRate / this.inputSampleRate)); const output = new Float32Array(outputLength); for (let i = 0; i < outputLength; i++) { const inputIndex = (i * this.inputSampleRate / this.outputSampleRate); const index = Math.floor(inputIndex); const fraction = inputIndex - index; const a = inputBuffer[index] || 0; const b = inputBuffer[index + 1] || 0; output[i] = a + fraction * (b - a); } return output; } float32ToPCM16(float32Array) { const pcm16 = new Int16Array(float32Array.length); for (let i = 0; i < float32Array.length; i++) { const sample = Math.max(-1, Math.min(1, float32Array[i])); pcm16[i] = Math.round(sample * 32767); } return pcm16; } sendAudioFrame(samples) { const pcm16Samples = this.float32ToPCM16(samples); this.audioSource.onData({ samples: pcm16Samples, sampleRate: this.outputSampleRate, channelCount: 1, bitsPerSample: 16 }); } async processAudioBuffer() { if (!this.isPlaying || this.audioBuffer.length < this.samplesPerFrame) { return; } const now = Date.now(); const timeSinceLastFrame = now - this.lastPlayTime; if (timeSinceLastFrame >= this.frameDuration) { const frame = this.audioBuffer.slice(0, this.samplesPerFrame); this.sendAudioFrame(frame); this.audioBuffer = this.audioBuffer.slice(this.samplesPerFrame); this.lastPlayTime = now; } // 调度下一帧 setTimeout(() => this.processAudioBuffer(), Math.max(0, this.frameDuration - timeSinceLastFrame)); } setPeerConnection(pc) { this.pc = pc; } reset() { console.log("AudioStreamer reset()"); this.lastPlayTime = 0; this.audioBuffer = new Float32Array(0); } cleanup() { this.isPlaying = false; if (this.track) { this.track.stop(); } } } module.exports = AudioStreamer;
解决方案
针对播放不稳定、出现机械音的问题,从时间同步、缓冲区管理、重采样精度三个核心方向修复:
1. 修复时间同步逻辑
原setTimeout延迟计算存在误差,导致帧发送节奏混乱,改用累计时间精确调度:
processAudioBuffer() { if (!this.isPlaying || this.audioBuffer.length < this.samplesPerFrame) { // 缓冲区不足时缩短检查间隔,避免漏帧 setTimeout(() => this.processAudioBuffer(), 1); return; } const now = Date.now(); const elapsed = now - this.lastPlayTime; // 计算理论上应发送的帧数,避免延迟累积 const framesToSend = Math.floor(elapsed / this.frameDuration); if (framesToSend > 0) { let sentFrames = 0; while (sentFrames < framesToSend && this.audioBuffer.length >= this.samplesPerFrame) { const frame = this.audioBuffer.slice(0, this.samplesPerFrame); this.sendAudioFrame(frame); this.audioBuffer = this.audioBuffer.slice(this.samplesPerFrame); sentFrames++; } // 更新为理论发送时间,避免时间跳变 this.lastPlayTime += framesToSend * this.frameDuration; } // 固定间隔调度,保证稳定的帧发送节奏 setTimeout(() => this.processAudioBuffer(), this.frameDuration); }
2. 优化缓冲区拼接效率
原代码每次拼接都创建新Float32Array,频繁内存操作易导致卡顿,改用动态数组管理:
constructor() { // ...其他属性 this.audioBuffer = []; // 用数组存储音频块,避免频繁扩容 this.bufferTotalLength = 0; // 跟踪总采样点数量 } handleLLMAudio(base64Audio) { // ...解析和重采样逻辑 const resampled = this.resampleBuffer(pcm16); // 添加到缓冲区数组 this.audioBuffer.push(resampled); this.bufferTotalLength += resampled.length; if (!this.isPlaying) { this.isPlaying = true; this.lastPlayTime = Date.now(); this.processAudioBuffer(); } } // 在processAudioBuffer中提取帧时合并足够的音频块: processAudioBuffer() { if (!this.isPlaying || this.bufferTotalLength < this.samplesPerFrame) { setTimeout(() => this.processAudioBuffer(), 1); return; } // 合并音频块直到有足够采样点 let merged = new Float32Array(0); while (this.bufferTotalLength >= this.samplesPerFrame && this.audioBuffer.length > 0) { const chunk = this.audioBuffer.shift(); const temp = new Float32Array(merged.length + chunk.length); temp.set(merged); temp.set(chunk, merged.length); merged = temp; this.bufferTotalLength -= chunk.length; if (merged.length >= this.samplesPerFrame) break; } // 提取单帧数据 const frame = merged.slice(0, this.samplesPerFrame); this.sendAudioFrame(frame); // 剩余采样点放回缓冲区 if (merged.length > this.samplesPerFrame) { this.audioBuffer.unshift(merged.slice(this.samplesPerFrame)); this.bufferTotalLength += merged.length - this.samplesPerFrame; } // 更新时间并调度下一帧 const now = Date.now(); const elapsed = now - this.lastPlayTime; this.lastPlayTime = elapsed >= this.frameDuration ? now : this.lastPlayTime + this.frameDuration; setTimeout(() => this.processAudioBuffer(), Math.max(0, this.lastPlayTime - now)); }
3. 提升重采样精度
原线性重采样的边界处理易产生失真,优化越界判断:
resampleBuffer(inputBuffer) { const outputLength = Math.round(inputBuffer.length * (this.outputSampleRate / this.inputSampleRate)); const output = new Float32Array(outputLength); const ratio = inputBuffer.length / outputLength; for (let i = 0; i < outputLength; i++) { const inputIndex = i * ratio; const index = Math.floor(inputIndex); const fraction = inputIndex - index; // 最后一个采样点直接取原值,避免越界 if (index >= inputBuffer.length - 1) { output[i] = inputBuffer[inputBuffer.length - 1] || 0; } else { const a = inputBuffer[index]; const b = inputBuffer[index + 1]; output[i] = a + fraction * (b - a); } } return output; }
4. 完善状态管理
在重置和清理方法中补充状态重置,避免残留状态影响后续播放:
reset() { console.log("AudioStreamer reset()"); this.lastPlayTime = 0; this.audioBuffer = []; this.bufferTotalLength = 0; this.isPlaying = false; } cleanup() { this.isPlaying = false; if (this.track) { this.track.stop(); this.track = null; } this.audioSource = null; this.mediaStream = null; this.reset(); }
内容的提问来源于stack exchange,提问作者Adam B
相关产品推荐
相关产品推荐

