You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Node.js处理超25MB音视频文件的OpenAI Whisper转录?

解决Whisper处理大文件的方案:音频压缩/分割 + Node.js视频处理

一、浏览器端音频压缩方案(替代lamejs)

用ffmpeg.wasm在浏览器中直接压缩音频,通过调整比特率、采样率减小文件体积,兼容性和稳定性优于lamejs。

代码示例:

import { createFFmpeg, fetchFile } from '@ffmpeg/ffmpeg';

const ffmpeg = createFFmpeg({ log: true });

async function compressAudio(file) {
  await ffmpeg.load();
  
  // 将原始文件写入ffmpeg内存
  ffmpeg.FS('writeFile', 'input.mp3', await fetchFile(file));
  
  // 压缩参数:64kbps比特率、44100Hz采样率(可按需调整)
  await ffmpeg.run('-i', 'input.mp3', '-b:a', '64k', '-ar', '44100', 'output.mp3');
  
  // 读取压缩后的文件并转为Blob
  const data = ffmpeg.FS('readFile', 'output.mp3');
  const compressedBlob = new Blob([data.buffer], { type: 'audio/mpeg' });
  
  // 清理内存中的临时文件
  ffmpeg.FS('unlink', 'input.mp3');
  ffmpeg.FS('unlink', 'output.mp3');
  
  return compressedBlob;
}

// 使用方式
// const compressedFile = await compressAudio(originalAudioFile);
// 提交compressedFile至Whisper API

注意:首次加载ffmpeg.wasm会下载核心库,建议添加加载状态提示;比特率越低体积越小,需在音质和文件大小间做平衡。

二、浏览器端音频分割方案

若压缩后仍超过25MB,可将音频分割为多个小片段,分别调用Whisper API后合并转录结果。

代码示例(基于Web Audio API):

async function splitAudio(file, chunkDurationMs = 300000) { // 默认按5分钟分割
  const audioContext = new (window.AudioContext || window.webkitAudioContext)();
  const arrayBuffer = await file.arrayBuffer();
  const audioBuffer = await audioContext.decodeAudioData(arrayBuffer);
  
  const chunks = [];
  const sampleRate = audioBuffer.sampleRate;
  const chunkSamples = Math.floor(chunkDurationMs / 1000 * sampleRate);
  
  for (let i = 0; i < audioBuffer.length; i += chunkSamples) {
    const end = Math.min(i + chunkSamples, audioBuffer.length);
    const chunkBuffer = audioContext.createBuffer(audioBuffer.numberOfChannels, end - i, sampleRate);
    
    // 复制音频数据到片段Buffer
    for (let channel = 0; channel < audioBuffer.numberOfChannels; channel++) {
      chunkBuffer.copyToChannel(audioBuffer.getChannelData(channel).slice(i, end), channel);
    }
    
    // 将Buffer转为WAV格式Blob
    const blob = await audioBufferToBlob(chunkBuffer);
    chunks.push(new File([blob], `chunk-${Math.floor(i/sampleRate)}.wav`, { type: 'audio/wav' }));
  }
  
  audioContext.close();
  return chunks;
}

// 辅助函数:AudioBuffer转WAV Blob
async function audioBufferToBlob(buffer) {
  const numberOfChannels = buffer.numberOfChannels;
  const length = buffer.length * numberOfChannels * 2 + 44;
  const arrayBuffer = new ArrayBuffer(length);
  const view = new DataView(arrayBuffer);
  
  // 写入WAV文件头
  const setUint16 = (offset, value) => view.setUint16(offset, value, true);
  const setUint32 = (offset, value) => view.setUint32(offset, value, true);
  
  setUint32(0, 0x46464952); // "RIFF"
  setUint32(4, length - 8); // 文件总长度-8
  setUint32(8, 0x45564157); // "WAVE"
  setUint32(12, 0x20746d66); // "fmt "
  setUint32(16, 16); // PCM格式长度
  setUint16(20, 1); // PCM编码
  setUint16(22, numberOfChannels);
  setUint32(24, buffer.sampleRate);
  setUint32(28, buffer.sampleRate * 2 * numberOfChannels); // 字节率
  setUint16(32, numberOfChannels * 2); // 块对齐
  setUint16(34, 16); // 位深度
  setUint32(36, 0x61746164); // "data"
  setUint32(40, length - 44); // 音频数据长度
  
  // 写入音频采样数据
  let offset = 44;
  for (let channel = 0; channel < numberOfChannels; channel++) {
    const channelData = buffer.getChannelData(channel);
    for (let i = 0; i < channelData.length; i++) {
      const sample = Math.max(-1, Math.min(1, channelData[i]));
      view.setInt16(offset, sample < 0 ? sample * 0x8000 : sample * 0x7FFF, true);
      offset += 2;
    }
  }
  
  return new Blob([arrayBuffer], { type: 'audio/wav' });
}

// 使用方式
// const audioChunks = await splitAudio(originalAudioFile);
// 遍历chunks调用Whisper API,最后拼接转录文本

三、Node.js处理大视频文件的Whisper转录方案

Node.js中先通过FFmpeg提取视频音频,再对音频压缩/分割,最后调用Whisper API完成转录。

前置依赖:

安装依赖包:

npm install fluent-ffmpeg openai

确保系统已安装FFmpeg,可通过ffmpeg -version验证

代码示例:

const ffmpeg = require('fluent-ffmpeg');
const { OpenAI } = require('openai');
const fs = require('fs');
const path = require('path');

const openai = new OpenAI({ apiKey: 'YOUR_API_KEY' });

// 从视频提取并压缩音频
async function extractAndCompressAudio(videoPath, outputAudioPath) {
  return new Promise((resolve, reject) => {
    ffmpeg(videoPath)
      .outputOptions('-b:a', '64k') // 音频比特率
      .outputOptions('-ar', '44100') // 采样率
      .save(outputAudioPath)
      .on('end', resolve)
      .on('error', reject);
  });
}

// 分割音频文件(若压缩后仍超25MB)
async function splitAudioNode(audioPath, chunkDir, chunkDurationSec = 300) {
  if (!fs.existsSync(chunkDir)) fs.mkdirSync(chunkDir);
  
  return new Promise((resolve, reject) => {
    ffmpeg(audioPath)
      .output(path.join(chunkDir, 'chunk_%03d.wav'))
      .outputOptions('-f', 'segment')
      .outputOptions('-segment_time', chunkDurationSec.toString())
      .outputOptions('-c', 'copy')
      .on('end', () => {
        const chunks = fs.readdirSync(chunkDir).map(file => path.join(chunkDir, file));
        resolve(chunks);
      })
      .on('error', reject)
      .run();
  });
}

// 调用Whisper API转录所有片段
async function transcribeChunks(chunks) {
  let fullTranscript = '';
  for (const chunkPath of chunks) {
    const transcription = await openai.audio.transcriptions.create({
      file: fs.createReadStream(chunkPath),
      model: 'whisper-1',
      language: 'zh' // 根据实际语言调整
    });
    fullTranscript += transcription.text + ' ';
    // 删除临时片段文件
    fs.unlinkSync(chunkPath);
  }
  return fullTranscript.trim();
}

// 完整处理流程
async function processLargeVideo(videoPath) {
  const tempAudioPath = './temp_audio.mp3';
  const tempChunkDir = './temp_chunks';
  
  try {
    // 提取压缩音频
    await extractAndCompressAudio(videoPath, tempAudioPath);
    
    // 检查文件大小
    const stats = fs.statSync(tempAudioPath);
    const fileSizeMB = stats.size / (1024 * 1024);
    
    let chunks = fileSizeMB > 25 
      ? await splitAudioNode(tempAudioPath, tempChunkDir) 
      : [tempAudioPath];
    
    // 转录并合并结果
    const fullTranscript = await transcribeChunks(chunks);
    console.log('完整转录结果:', fullTranscript);
    
    // 清理临时文件
    fs.unlinkSync(tempAudioPath);
    if (fs.existsSync(tempChunkDir)) fs.rmdirSync(tempChunkDir);
    
    return fullTranscript;
  } catch (err) {
    console.error('处理失败:', err);
    // 异常时清理临时文件
    if (fs.existsSync(tempAudioPath)) fs.unlinkSync(tempAudioPath);
    if (fs.existsSync(tempChunkDir)) fs.rmdirSync(tempChunkDir, { recursive: true });
    throw err;
  }
}

// 使用方式
// processLargeVideo('./large_video.mp4');

内容的提问来源于stack exchange,提问作者Snehal Shyamsukha

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.11 14:37:33