You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何通过MediaRecorder或JS媒体API实现重叠时移帧音频录制?

实现滑动窗口式重叠音频录制(Web Audio API方案)

MediaRecorder API确实做不到这种重叠片段的录制——它的设计逻辑是每次start()到stop()生成一段独立的、无重叠的编码后音频片段,没有提供访问历史录制数据的接口。要实现你需要的2秒窗口+0.1秒跳步的滑动录制,得用Web Audio API直接处理原始音频数据,自己管理缓存和窗口截取。

核心思路

  1. 捕获麦克风的原始音频流,通过Web Audio API获取PCM格式的原始音频样本。
  2. 维护一个缓存容器,始终存储最近2秒的音频样本。
  3. 每隔0.1秒,从缓存中提取完整的2秒样本,编码成后端需要的格式(比如WAV),然后发送到服务器。

具体实现步骤

1. 初始化音频上下文与媒体流

首先获取麦克风权限,创建AudioContext并连接媒体源:

async function initAudio() {
  const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
  const audioContext = new AudioContext({ sampleRate: 16000 }); // 匹配后端模型的采样率
  const source = audioContext.createMediaStreamSource(stream);
  
  // 使用AudioWorklet处理音频(替代已废弃的ScriptProcessorNode)
  await audioContext.audioWorklet.addModule('audio-processor.js');
  const processor = new AudioWorkletNode(audioContext, 'sliding-window-processor');
  
  // 传递配置参数给Worklet:窗口大小(2秒)、跳步时间(0.1秒)
  processor.port.postMessage({
    windowSize: audioContext.sampleRate * 2,
    hopSize: audioContext.sampleRate * 0.1,
    sampleRate: audioContext.sampleRate
  });
  
  // 监听Worklet发送的完整窗口数据
  processor.port.onmessage = (e) => {
    if (e.data.type === 'windowData') {
      sendToBackend(e.data.pcmData);
    }
  };
  
  source.connect(processor);
  processor.connect(audioContext.destination);
}

2. 编写AudioWorklet处理器(audio-processor.js)

在独立的Worklet文件中处理音频流,维护缓存并生成滑动窗口:

class SlidingWindowProcessor extends AudioWorkletProcessor {
  constructor() {
    super();
    this.buffer = []; // 存储原始PCM样本
    this.windowSize = 0;
    this.hopSize = 0;
    this.sampleRate = 0;
    this.hopCounter = 0;
    
    // 接收主线程的配置
    this.port.onmessage = (e) => {
      this.windowSize = e.data.windowSize;
      this.hopSize = e.data.hopSize;
      this.sampleRate = e.data.sampleRate;
    };
  }

  process(inputs, outputs) {
    const input = inputs[0];
    if (input.length === 0) return true;
    
    // 将输入的Float32Array样本转成普通数组,追加到缓存
    const newSamples = Array.from(input[0]);
    this.buffer = [...this.buffer, ...newSamples];
    
    // 裁剪缓存,只保留最近windowSize个样本(2秒)
    if (this.buffer.length > this.windowSize) {
      this.buffer = this.buffer.slice(this.buffer.length - this.windowSize);
    }
    
    // 每积累hopSize个样本,就发送一次完整窗口数据
    this.hopCounter += input[0].length;
    if (this.hopCounter >= this.hopSize && this.buffer.length === this.windowSize) {
      this.port.postMessage({
        type: 'windowData',
        pcmData: new Float32Array(this.buffer)
      });
      this.hopCounter -= this.hopSize;
    }
    
    return true;
  }
}

registerProcessor('sliding-window-processor', SlidingWindowProcessor);

3. 将PCM数据编码为可发送的Blob

后端通常需要编码后的音频格式(比如WAV),这里提供一个简单的PCM转WAV的函数:

function pcmToWav(pcmData, sampleRate, channels = 1) {
  const buffer = new ArrayBuffer(44 + pcmData.length * 2);
  const view = new DataView(buffer);
  
  // 写入WAV文件头
  view.setUint8(0, 0x52); view.setUint8(1, 0x49); view.setUint8(2, 0x46); view.setUint8(3, 0x46); // RIFF
  view.setUint32(4, 36 + pcmData.length * 2, true);
  view.setUint8(8, 0x57); view.setUint8(9, 0x41); view.setUint8(10, 0x56); view.setUint8(11, 0x45); // WAVE
  view.setUint8(12, 0x66); view.setUint8(13, 0x6D); view.setUint8(14, 0x74); view.setUint8(15, 0x20); // fmt
  view.setUint32(16, 16, true); // PCM格式长度
  view.setUint16(20, 1, true); // PCM编码
  view.setUint16(22, channels, true);
  view.setUint32(24, sampleRate, true);
  view.setUint32(28, sampleRate * channels * 2, true); // 比特率
  view.setUint16(32, channels * 2, true); // 块对齐
  view.setUint16(34, 16, true); // 位深度
  view.setUint8(36, 0x64); view.setUint8(37, 0x61); view.setUint8(38, 0x74); view.setUint8(39, 0x61); // data
  view.setUint32(40, pcmData.length * 2, true);
  
  // 写入PCM数据(Float32转16位整数)
  let offset = 44;
  for (const sample of pcmData) {
    const intSample = Math.max(-32768, Math.min(32767, sample * 32767));
    view.setInt16(offset, intSample, true);
    offset += 2;
  }
  
  return new Blob([buffer], { type: 'audio/wav' });
}

// 发送到后端的函数
async function sendToBackend(pcmData) {
  const wavBlob = pcmToWav(pcmData, 16000);
  const formData = new FormData();
  formData.append('audio', wavBlob, 'window.wav');
  
  try {
    await fetch('/api/predict', {
      method: 'POST',
      body: formData
    });
  } catch (err) {
    console.error('发送音频失败:', err);
  }
}

关键注意事项

  • 采样率一致性:确保AudioContext的采样率和后端模型要求的一致(比如常见的16000Hz),否则模型无法正常处理。
  • 编码性能:如果需要MP3等压缩格式,纯JS编码可能性能不足,可以用WebAssembly库来加速编码。
  • 内存管理:缓存只保留最近2秒的数据,避免内存持续增长;如果长时间运行,可定期清理无效数据。
  • 权限问题:需要在HTTPS环境下才能获取麦克风权限(本地开发的localhost除外)。

内容的提问来源于stack exchange,提问作者miccio

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 14:00:48