You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

NodeJS通过WebRTC转发GPT-4o-audio-preview音频流异常排查

问题:NodeJS通过WebRTC转发GPT-4o-audio-preview音频流播放不稳定

我在NodeJS应用里尝试把OpenAI的gpt-4o-audio-preview模型生成的音频通过WebRTC(使用flutter_webrtc)转发给客户端,为此编写了AudioStreamer类,负责转换模型返回的音频delta、将采样率从24kHz上采样至48kHz,并把缓冲区拆分为接近实时的流。但音频播放效果不稳定:有时能完整正常播放,有时仅能听到开头几个词,随后迅速变为机械音并停止。

handleLLMAudio方法直接接收OpenAI库的输出,传入的是event.choices[0]?.delta?.audio?.data的值。

我的AudioStreamer类:

const wrtc = require("@roamhq/wrtc");
    
class AudioStreamer {
   constructor() {
      this.pc = null;
      this.inputSampleRate = 24000; // gpt-4o-audio-preview输出采样率
      this.outputSampleRate = 48000; // WebRTC采样率
      this.samplesPerFrame = 480;  // PCM16格式下,960字节=480个采样点
      this.audioBuffer = new Float32Array(0);
      this.track = null;
      this.mediaStream = null;
      this.isPlaying = false;
      this.lastPlayTime = 0;
      
      // 计算每帧的持续时间(毫秒)
      this.frameDuration = (this.samplesPerFrame / this.outputSampleRate) * 1000; // ~10ms
   }

   /**
    * 新的RTCPeerConnection和RTCAudioSource创建后调用
    * 调用此方法前会先执行this.setPeerConnection
    * @param {RTCAudioSource} audioSource 
    */
   async initialize(audioSource) {
      const { MediaStream } = wrtc;
      this.audioSource = audioSource;
      this.track = this.audioSource.createTrack();
      this.pc.addTrack(this.track);
      this.mediaStream = new MediaStream([this.audioSource]);
   }

   /**
    * 接收base64格式的音频字符串,开始转换为实时音频流
    * @param {String} base64Audio 直接来自gpt-4o-audio-preview的PCM16音频delta
    */
   handleLLMAudio(base64Audio) {
      const buffer = Buffer.from(base64Audio, 'base64');
      const view = new DataView(buffer.buffer, buffer.byteOffset, buffer.length);
      const pcm16 = new Float32Array(buffer.length / 2);
      
      for (let i = 0; i < pcm16.length; i++) {
         const int16Value = view.getInt16(i * 2, true);
         pcm16[i] = int16Value / 32768.0;
      }

      const resampled = this.resampleBuffer(pcm16);

      // 添加到缓冲区
      const newBuffer = new Float32Array(this.audioBuffer.length + resampled.length);
      newBuffer.set(this.audioBuffer);
      newBuffer.set(resampled, this.audioBuffer.length);
      this.audioBuffer = newBuffer;

      // 如果未开始播放则启动
      if (!this.isPlaying) {
         this.isPlaying = true;
         this.lastPlayTime = Date.now();
         this.processAudioBuffer();
      }
   }

   resampleBuffer(inputBuffer) {
      const outputLength = Math.ceil(inputBuffer.length * (this.outputSampleRate / this.inputSampleRate));
      const output = new Float32Array(outputLength);
      
      for (let i = 0; i < outputLength; i++) {
         const inputIndex = (i * this.inputSampleRate / this.outputSampleRate);
         const index = Math.floor(inputIndex);
         const fraction = inputIndex - index;
         
         const a = inputBuffer[index] || 0;
         const b = inputBuffer[index + 1] || 0;
         output[i] = a + fraction * (b - a);
      }
      
      return output;
   }

   float32ToPCM16(float32Array) {
      const pcm16 = new Int16Array(float32Array.length);
      for (let i = 0; i < float32Array.length; i++) {
         const sample = Math.max(-1, Math.min(1, float32Array[i]));
         pcm16[i] = Math.round(sample * 32767);
      }
      return pcm16;
   }

   sendAudioFrame(samples) {
      const pcm16Samples = this.float32ToPCM16(samples);
      
      this.audioSource.onData({
         samples: pcm16Samples,
         sampleRate: this.outputSampleRate,
         channelCount: 1,
         bitsPerSample: 16
      });
   }

   async processAudioBuffer() {
      if (!this.isPlaying || this.audioBuffer.length < this.samplesPerFrame) {
         return;
      }

      const now = Date.now();
      const timeSinceLastFrame = now - this.lastPlayTime;

      if (timeSinceLastFrame >= this.frameDuration) {
         const frame = this.audioBuffer.slice(0, this.samplesPerFrame);
         this.sendAudioFrame(frame);
         this.audioBuffer = this.audioBuffer.slice(this.samplesPerFrame);
         this.lastPlayTime = now;
      }

      // 调度下一帧
      setTimeout(() => this.processAudioBuffer(), Math.max(0, this.frameDuration - timeSinceLastFrame));
   }

   setPeerConnection(pc) {
      this.pc = pc;
   }

   reset() {
      console.log("AudioStreamer reset()");
      this.lastPlayTime = 0;
      this.audioBuffer = new Float32Array(0);
   }

   cleanup() {
      this.isPlaying = false;
      if (this.track) {
         this.track.stop();
      }
   }
}

module.exports = AudioStreamer;
解决方案

针对播放不稳定、出现机械音的问题,从时间同步、缓冲区管理、重采样精度三个核心方向修复:

1. 修复时间同步逻辑

原setTimeout延迟计算存在误差,导致帧发送节奏混乱,改用累计时间精确调度:

processAudioBuffer() {
  if (!this.isPlaying || this.audioBuffer.length < this.samplesPerFrame) {
    // 缓冲区不足时缩短检查间隔,避免漏帧
    setTimeout(() => this.processAudioBuffer(), 1);
    return;
  }

  const now = Date.now();
  const elapsed = now - this.lastPlayTime;
  // 计算理论上应发送的帧数,避免延迟累积
  const framesToSend = Math.floor(elapsed / this.frameDuration);

  if (framesToSend > 0) {
    let sentFrames = 0;
    while (sentFrames < framesToSend && this.audioBuffer.length >= this.samplesPerFrame) {
      const frame = this.audioBuffer.slice(0, this.samplesPerFrame);
      this.sendAudioFrame(frame);
      this.audioBuffer = this.audioBuffer.slice(this.samplesPerFrame);
      sentFrames++;
    }
    // 更新为理论发送时间,避免时间跳变
    this.lastPlayTime += framesToSend * this.frameDuration;
  }

  // 固定间隔调度,保证稳定的帧发送节奏
  setTimeout(() => this.processAudioBuffer(), this.frameDuration);
}

2. 优化缓冲区拼接效率

原代码每次拼接都创建新Float32Array,频繁内存操作易导致卡顿,改用动态数组管理:

constructor() {
  // ...其他属性
  this.audioBuffer = []; // 用数组存储音频块,避免频繁扩容
  this.bufferTotalLength = 0; // 跟踪总采样点数量
}

handleLLMAudio(base64Audio) {
  // ...解析和重采样逻辑
  const resampled = this.resampleBuffer(pcm16);
  
  // 添加到缓冲区数组
  this.audioBuffer.push(resampled);
  this.bufferTotalLength += resampled.length;

  if (!this.isPlaying) {
    this.isPlaying = true;
    this.lastPlayTime = Date.now();
    this.processAudioBuffer();
  }
}

// 在processAudioBuffer中提取帧时合并足够的音频块:
processAudioBuffer() {
  if (!this.isPlaying || this.bufferTotalLength < this.samplesPerFrame) {
    setTimeout(() => this.processAudioBuffer(), 1);
    return;
  }

  // 合并音频块直到有足够采样点
  let merged = new Float32Array(0);
  while (this.bufferTotalLength >= this.samplesPerFrame && this.audioBuffer.length > 0) {
    const chunk = this.audioBuffer.shift();
    const temp = new Float32Array(merged.length + chunk.length);
    temp.set(merged);
    temp.set(chunk, merged.length);
    merged = temp;
    this.bufferTotalLength -= chunk.length;
    if (merged.length >= this.samplesPerFrame) break;
  }

  // 提取单帧数据
  const frame = merged.slice(0, this.samplesPerFrame);
  this.sendAudioFrame(frame);
  // 剩余采样点放回缓冲区
  if (merged.length > this.samplesPerFrame) {
    this.audioBuffer.unshift(merged.slice(this.samplesPerFrame));
    this.bufferTotalLength += merged.length - this.samplesPerFrame;
  }

  // 更新时间并调度下一帧
  const now = Date.now();
  const elapsed = now - this.lastPlayTime;
  this.lastPlayTime = elapsed >= this.frameDuration ? now : this.lastPlayTime + this.frameDuration;
  setTimeout(() => this.processAudioBuffer(), Math.max(0, this.lastPlayTime - now));
}

3. 提升重采样精度

原线性重采样的边界处理易产生失真,优化越界判断:

resampleBuffer(inputBuffer) {
  const outputLength = Math.round(inputBuffer.length * (this.outputSampleRate / this.inputSampleRate));
  const output = new Float32Array(outputLength);
  const ratio = inputBuffer.length / outputLength;

  for (let i = 0; i < outputLength; i++) {
    const inputIndex = i * ratio;
    const index = Math.floor(inputIndex);
    const fraction = inputIndex - index;

    // 最后一个采样点直接取原值,避免越界
    if (index >= inputBuffer.length - 1) {
      output[i] = inputBuffer[inputBuffer.length - 1] || 0;
    } else {
      const a = inputBuffer[index];
      const b = inputBuffer[index + 1];
      output[i] = a + fraction * (b - a);
    }
  }
  return output;
}

4. 完善状态管理

在重置和清理方法中补充状态重置,避免残留状态影响后续播放:

reset() {
  console.log("AudioStreamer reset()");
  this.lastPlayTime = 0;
  this.audioBuffer = [];
  this.bufferTotalLength = 0;
  this.isPlaying = false;
}

cleanup() {
  this.isPlaying = false;
  if (this.track) {
    this.track.stop();
    this.track = null;
  }
  this.audioSource = null;
  this.mediaStream = null;
  this.reset();
}

内容的提问来源于stack exchange,提问作者Adam B

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.14 12:48:11