You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用OpenAI Whisper API实现浏览器录音转写时遇400错误求助

浏览器录制的MP3调用OpenAI Whisper API返回400错误,本地录制文件正常

我基于Node.js和React开发录音转写工具,用户在浏览器录制音频后转成MP3,通过fs.createReadStream读取文件传入Whisper API的createTranscription接口时返回400错误,但用Windows录音机录制的MP3文件执行相同操作却能正常调用API。已确认保存的MP3可正常播放、API密钥正确,尝试过修改文件保存方式、多种buffer格式转换、直接上传buffer等方案均无效。

相关代码

React录音组件

import React, { useState, useEffect, useRef } from "react";

import Microphone from "./Microphone/Microphone";
const TSST = () => {
  const BASE_URL = process.env.REACT_APP_SERVER_URL || "http://localhost:5000";

  const mediaRecorder = useRef(null);
  const [stream, setStream] = useState(null);
  const [audioChunks, setAudioChunks] = useState([]);
  const [audio, setAudio] = useState(null);
  const [audioFile, setAudioFile] = useState(null);
  const [transcribtion, setTranscription] = useState("");
  const [audioBlob, setAudioBlob] = useState("");
  const [audioBuffer, setAudioBuffer] = useState("");

  useEffect(() => {
    const initializeMediaRecorder = async () => {
      if ("MediaRecorder" in window) {
        try {
            const streamData = await navigator.mediaDevices.getUserMedia({ audio: true });
            setStream(streamData);
        } catch (err) {
            console.log(err.message);
        }
      } else {
          console.log("The MediaRecorder API is not supported in your browser.");
      }
    }

    initializeMediaRecorder();
  }, [])

  const handleStartRecording = () => {
    const media = new MediaRecorder(stream, { type: "audio/mp3" });

    mediaRecorder.current = media;
    mediaRecorder.current.start();

    let chunks = [];
    mediaRecorder.current.ondataavailable = (e) => {
       chunks.push(e.data);
    };
    setAudioChunks(chunks);
  }
  const handleStopRecording = () => {
    mediaRecorder.current.stop();
    mediaRecorder.current.onstop = () => {
      const audioBlob = new Blob(audioChunks, { type: "audio/mp3" });
      const audioUrl = URL.createObjectURL(audioBlob);

      setAudioBlob(audioBlob)
      setAudio(audioUrl);
      setAudioChunks([]);

      let file = new File([audioUrl], "recorded_audio.mp3",{type:"audio/mp3", lastModified:new Date().getTime()});
      let container = new DataTransfer();
      container.items.add(file);
      document.getElementById("audioFile").files = container.files;
      setAudioFile(container.files[0]);

      console.log(file);
    };
  }

  const handleSubmitRecording = async () => {
    try {
      const reader = new FileReader();
      reader.onloadend = async () => {
        const base64String = reader.result.split(',')[1];
        const res = await fetch(`${BASE_URL}/api/openai/transcriber`, {
          method: "POST",
          headers: {
            "Content-Type": "application/json",
          },
          body: JSON.stringify({ audioBuffer: base64String, lang: "en" })
        })
        const data = await res.json();
        setTranscription(data);
      };
      reader.readAsDataURL(audioBlob);

    } catch (error) {
      console.log(error);
    } finally {
    }
  }

    return (
      <div className="h-[calc(100vh-73px)] flex justify-center items-center">
        <div className="w-[40%] flex justify-between items-center">
          <div className="flex flex-col">
            <Microphone startFunction={ handleStartRecording } stopFunction={ handleStopRecording } />
            <button onClick={handleStartRecording} className="w-fit my-10 p-5 bg-gray-200 rounded-lg">Start Recording</button>
            <button onClick={handleStopRecording} className="w-fit mb-10 p-5 bg-gray-200 rounded-lg">Stop Recording</button>

            <audio className="mb-10" src={audio && audio} controls></audio>
            <input id="audioFile" type="file" onChange={ (e) => {setAudioFile(e.target.files[0])}}/>
          </div>
          
          <div>
            <button className="p-10 bg-yellow-500 rounded-xl" onClick={ handleSubmitRecording } >Submit</button>
          </div>
        </div>

        <div className="w-[40%] flex justify-center items-center">
          <textarea value={transcribtion} readOnly className="w-[60%] aspect-square resize-none shadow-lg shadow-black"></textarea>
        </div>
      </div>
    );
};
export default TSST;

API接口代码

export const transcribe = async (req, res) => {
    const { audioBuffer, lang} = req.body;

    const audioBufferBase64 = Buffer.from(audioBuffer, 'base64');

    const fileName = "test.mp3";
    const folderName = `./audio/${fileName}`

    const writableStream = fs.createWriteStream(folderName);
    writableStream.write(audioBufferBase64);

    const readStream = fs.createReadStream(folderName);

    readStream.on('data', (data) => {
        console.log('Read stream data:', data);
    });

    try {
        const whisperRes = await openai.createTranscription(
            readStream,
            "whisper-1",
        )

        const chatResponse = whisperRes.data.text;
        console.log(chatResponse)

        res.status(200).json({ chatResponse: chatResponse });
    } catch (error) {
        res.status(500).json({ message: error });
    }
}

服务端代码

import express from "express";
import cors from "cors";
import * as dotenv from "dotenv";
import mongoose from "mongoose";
import multer from "multer";

import { dalle, chatGPT, summarize, translate, transcribe } from "./api/openai.js";
import { getImages, postImage } from "./api/imageShowcase.js";
import { login, signup } from "./api/user.js";

dotenv.config();

const app = express();
const upload = multer();
const storage = multer.memoryStorage();
const uploadMiddleware = multer({ storage: storage });

app.use(cors());
app.use(express.json({limit: '50mb'}));

const atlasURL = process.env.MONGODB_URL;    
const PORT = process.env.PORT || 5000;

mongoose.connect(atlasURL)
    .then(() => app.listen(PORT, () => console.log(`Successfully connected to port ${PORT}`)))
    .catch(error => console.log("There was an error: ", error));

app.get("/", async (req, res) => {
    res.send("Server is RUNNING");
})

app.post("/api/openai/transcriber",(req, res) => transcribe(req, res));

解决方案

1. 修复MediaRecorder格式问题

多数浏览器不支持原生MP3编码,即使指定type: "audio/mp3",实际输出的是其他格式(如webm),只是Blob的type被标记为MP3,导致Whisper API识别失败。

修改React录制代码,改用浏览器原生支持的WAV格式:

// 启动录制时指定WAV格式
const handleStartRecording = () => {
  const media = new MediaRecorder(stream, { type: "audio/wav" });
  // ... 其他代码不变
}

// 停止录制时生成WAV格式的Blob和File
const handleStopRecording = () => {
  mediaRecorder.current.stop();
  mediaRecorder.current.onstop = () => {
    const audioBlob = new Blob(audioChunks, { type: "audio/wav" });
    // ...
    let file = new File([audioBlob], "recorded_audio.wav",{type:"audio/wav", lastModified:new Date().getTime()});
    // ...
  };
}

2. 等待文件写入完成再读取

服务端中用createWriteStream写入文件后,立刻读取会导致文件不完整,需等待写入流结束后再操作:

export const transcribe = async (req, res) => {
    const { audioBuffer, lang} = req.body;
    const audioBufferBase64 = Buffer.from(audioBuffer, 'base64');
    const fileName = "test.wav";
    const folderName = `./audio/${fileName}`;

    // 用Promise等待写入完成
    await new Promise((resolve, reject) => {
        const writableStream = fs.createWriteStream(folderName);
        writableStream.write(audioBufferBase64);
        writableStream.end(); // 触发finish事件
        writableStream.on('finish', resolve);
        writableStream.on('error', reject);
    });

    const readStream = fs.createReadStream(folderName);

    try {
        const whisperRes = await openai.createTranscription(
            readStream,
            "whisper-1",
            undefined,
            "text",
            undefined,
            lang // 传入语言参数
        )
        res.status(200).json({ chatResponse: whisperRes.data.text });
    } catch (error) {
        console.error(error.response?.data || error.message); // 打印详细错误
        res.status(error.response?.status || 500).json({ message: error.response?.data?.error || error.message });
    }
}

3. 修复前端File创建错误

前端中new File([audioUrl], ...)是错误的,audioUrl是Object URL而非原始Blob数据,应直接用audioBlob创建File:

// 替换错误的File创建代码
let file = new File([audioBlob], "recorded_audio.wav",{type:"audio/wav", lastModified:new Date().getTime()});

4. 改用FormData传输优化(推荐)

放弃base64转JSON的方式,用FormData直接传文件,减少转换损耗和格式问题:

  • 前端修改提交逻辑:
const handleSubmitRecording = async () => {
    try {
        const formData = new FormData();
        formData.append('audio', audioBlob, 'recorded_audio.wav');
        formData.append('lang', 'en');

        const res = await fetch(`${BASE_URL}/api/openai/transcriber`, {
            method: "POST",
            body: formData
        })
        const data = await res.json();
        setTranscription(data.chatResponse);
    } catch (error) {
        console.log(error);
    }
}
  • 服务端修改路由和接口:
// 路由改用multer中间件接收文件
app.post("/api/openai/transcriber", uploadMiddleware.single('audio'), transcribe);

// 更新transcribe函数
export const transcribe = async (req, res) => {
    const lang = req.body.lang;
    const audioFile = req.file;

    try {
        // 直接传入buffer,无需写入本地文件
        const whisperRes = await openai.createTranscription(
            Buffer.from(audioFile.buffer),
            "whisper-1",
            undefined,
            "text",
            undefined,
            lang
        )
        res.status(200).json({ chatResponse: whisperRes.data.text });
    } catch (error) {
        console.error(error.response?.data || error.message);
        res.status(error.response?.status || 500).json({ message: error.response?.data?.error || error.message });
    }
}

内容的提问来源于stack exchange,提问作者Ali Jalloul

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.23 11:12:01