You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

网页录音分段上传API转写失败:首次成功后报格式不支持错误

录音分段转写:首次请求正常,后续返回400格式错误的解决方法

问题描述

我在网页实现了录音功能,希望每隔3秒发送POST请求到API做语音转写。首次请求能正常工作,但后续所有请求都返回400错误:

Error code: 400 - {'error': {'message': 'The audio file could not be decoded or its format is not supported.', 'type': 'invalid_request_error', 'param': None, 'code': None}}

试过使用临时文件名、调整请求间隔,问题仍未解决,相关代码如下:

let mediaRecorder;
let audioChunks = [];
let sendInterval;

document.getElementById('recordBtn').addEventListener('click', async () => {
    if (mediaRecorder && mediaRecorder.state === "recording") {
        mediaRecorder.stop();
        clearInterval(sendInterval); // Clear the interval when stopping
        document.getElementById('recordBtn').textContent = 'Start Recording';
    } else {
        try {
            const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
            mediaRecorder = new MediaRecorder(stream);
            mediaRecorder.ondataavailable = event => {
                if (event.data.size > 0) {
                    audioChunks.push(event.data);
                    console.log('Chunk received:', event.data.size);
                }
            };

            // Start recording with timeslice to ensure chunks are generated at regular intervals
            mediaRecorder.start(3000);
            document.getElementById('recordBtn').textContent = 'Stop Recording';
            var x = 1;

            const sendAudioChunks = async () => {
                if (audioChunks.length > 0) {
                    console.log('Sending chunks:', audioChunks.length);
                    const audioBlob = new Blob(audioChunks, { 'type': 'audio/wav' });
                    audioChunks = []; // Clear chunks after sending
                    const formData = new FormData();
                    formData.append('audio_file', audioBlob, 'audio.wav');

                    try {
                        // Inside your try block within sendAudioChunks
                        const response = await fetch('/transcribe', {
                            method: 'POST',
                            body: formData,
                        });
                        if (!response.ok) {
                            // Log or handle non-200 responses here
                            console.error('Server responded with non-200 status:', response.status);
                        }

                        const data = await response.json();
                        console.log('Server response:', data);
                        document.getElementById('transcriptionContainer').textContent = data.transcription || 'Transcription failed or was empty.';


                    } catch (error) {
                        console.error('Failed to send audio chunks:', error);
                    }
                } else {
                    console.log('No chunks to send');
                }
            };

            clearInterval(sendInterval);
            sendInterval = setInterval(sendAudioChunks, 3000);

            mediaRecorder.onstop = async () => {
                clearInterval(sendInterval); // Clear the interval when stopping
                await sendAudioChunks(); // Send any remaining chunks
            };
        } catch (error) {
            console.error('Error accessing media devices:', error);
        }
    }
});

问题根源

核心错误是强行将MediaRecorder生成的音频片段标记为WAV格式,但这些片段本身并不是完整的WAV文件。

MediaRecorder默认输出WebM或OGG格式(取决于浏览器),用start(3000)生成的chunk是这些格式的片段,缺少WAV文件必需的文件头信息。首次请求可能因为API容错性能被识别,但后续片段只有纯音频数据,没有格式标识头,导致API无法解码,返回400错误。

解决方案

有两种可行的修复方向:

方向1:使用MediaRecorder原生格式发送(推荐,无需转换)

如果API支持WebM/OGG格式,直接用MediaRecorder的实际输出格式发送即可:

let mediaRecorder;
let audioChunks = [];
let sendInterval;

document.getElementById('recordBtn').addEventListener('click', async () => {
    if (mediaRecorder && mediaRecorder.state === "recording") {
        mediaRecorder.stop();
        clearInterval(sendInterval);
        document.getElementById('recordBtn').textContent = 'Start Recording';
    } else {
        try {
            const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
            mediaRecorder = new MediaRecorder(stream);
            // 获取MediaRecorder实际使用的MIME类型
            const mimeType = mediaRecorder.mimeType || 'audio/webm';
            console.log('当前录音格式:', mimeType);

            mediaRecorder.ondataavailable = event => {
                if (event.data.size > 0) {
                    audioChunks.push(event.data);
                    console.log('收到音频片段:', event.data.size);
                }
            };

            mediaRecorder.start(3000);
            document.getElementById('recordBtn').textContent = '停止录音';

            const sendAudioChunks = async () => {
                if (audioChunks.length > 0) {
                    console.log('发送片段数量:', audioChunks.length);
                    // 使用原生格式创建Blob
                    const audioBlob = new Blob(audioChunks, { 'type': mimeType });
                    audioChunks = [];
                    const formData = new FormData();
                    // 文件名后缀和格式匹配
                    const fileName = `audio.${mimeType.split('/')[1]}`;
                    formData.append('audio_file', audioBlob, fileName);

                    try {
                        const response = await fetch('/transcribe', {
                            method: 'POST',
                            body: formData,
                        });
                        if (!response.ok) {
                            const errorDetails = await response.text();
                            console.error('服务器错误:', response.status, errorDetails);
                            return;
                        }

                        const data = await response.json();
                        console.log('转写结果:', data);
                        document.getElementById('transcriptionContainer').textContent = data.transcription || '无有效转写结果';
                    } catch (error) {
                        console.error('发送音频失败:', error);
                    }
                } else {
                    console.log('无音频片段可发送');
                }
            };

            clearInterval(sendInterval);
            sendInterval = setInterval(sendAudioChunks, 3000);

            mediaRecorder.onstop = async () => {
                clearInterval(sendInterval);
                await sendAudioChunks();
            };
        } catch (error) {
            console.error('访问麦克风失败:', error);
        }
    }
});

方向2:将片段转换为完整WAV格式(若API仅支持WAV)

如果API只接受WAV格式,用Web Audio API将MediaRecorder输出转换为带完整头的WAV文件:

首先添加转换函数:

async function blobToWav(blob) {
    const arrayBuffer = await blob.arrayBuffer();
    const audioContext = new (window.AudioContext || window.webkitAudioContext)();
    const audioBuffer = await audioContext.decodeAudioData(arrayBuffer);

    // 创建WAV文件头和数据
    const length = audioBuffer.length * audioBuffer.numberOfChannels * 2 + 44;
    const arrayBufferWav = new ArrayBuffer(length);
    const view = new DataView(arrayBufferWav);
    const channels = [];
    let offset = 0;
    let pos = 0;

    // 写入WAV头信息
    const setUint16 = (data) => {
        view.setUint16(pos, data, true);
        pos += 2;
    };
    const setUint32 = (data) => {
        view.setUint32(pos, data, true);
        pos += 4;
    };

    setUint32(0x46464952); // "RIFF"
    setUint32(length - 8); // 文件总长度-8
    setUint32(0x45564157); // "WAVE"
    setUint32(0x20746d66); // "fmt "
    setUint32(16); // PCM格式块大小
    setUint16(1); // PCM编码类型
    setUint16(audioBuffer.numberOfChannels); // 声道数
    setUint32(audioBuffer.sampleRate); // 采样率
    setUint32(audioBuffer.sampleRate * 2 * audioBuffer.numberOfChannels); // 字节率
    setUint16(audioBuffer.numberOfChannels * 2); // 块对齐
    setUint16(16); // 位深度
    setUint32(0x61746164); // "data"
    setUint32(length - pos - 4); // 音频数据长度

    // 写入音频采样数据
    for (let i = 0; i < audioBuffer.numberOfChannels; i++) {
        channels.push(audioBuffer.getChannelData(i));
    }

    while (pos < length) {
        for (let i = 0; i < audioBuffer.numberOfChannels; i++) {
            let sample = Math.max(-1, Math.min(1, channels[i][offset]));
            sample = sample < 0 ? sample * 0x8000 : sample * 0x7FFF;
            view.setInt16(pos, sample, true);
            pos += 2;
        }
        offset++;
    }

    return new Blob([arrayBufferWav], { type: 'audio/wav' });
}

然后在sendAudioChunks中调用转换:

const sendAudioChunks = async () => {
    if (audioChunks.length > 0) {
        console.log('发送片段数量:', audioChunks.length);
        const audioBlob = new Blob(audioChunks, { 'type': mediaRecorder.mimeType });
        audioChunks = [];
        // 转换为WAV格式
        const wavBlob = await blobToWav(audioBlob);
        const formData = new FormData();
        formData.append('audio_file', wavBlob, 'audio.wav');

        try {
            const response = await fetch('/transcribe', {
                method: 'POST',
                body: formData,
            });
            if (!response.ok) {
                const errorDetails = await response.text();
                console.error('服务器错误:', response.status, errorDetails);
                return;
            }

            const data = await response.json();
            console.log('转写结果:', data);
            document.getElementById('transcriptionContainer').textContent = data.transcription || '无有效转写结果';
        } catch (error) {
            console.error('发送音频失败:', error);
        }
    } else {
        console.log('无音频片段可发送');
    }
};

额外注意事项

  • 测试时可在控制台打印mediaRecorder.mimeType,确认当前录音格式
  • WAV转换会有一定性能开销,短片段影响不大,长片段建议优化
  • 若API支持流式转写,可考虑用WebSocket代替定时POST,效率更高

内容的提问来源于stack exchange,提问作者friso

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.27 19:17:18