网页录音分段上传API转写失败:首次成功后报格式不支持错误
录音分段转写:首次请求正常,后续返回400格式错误的解决方法
问题描述
我在网页实现了录音功能,希望每隔3秒发送POST请求到API做语音转写。首次请求能正常工作,但后续所有请求都返回400错误:
Error code: 400 - {'error': {'message': 'The audio file could not be decoded or its format is not supported.', 'type': 'invalid_request_error', 'param': None, 'code': None}}
试过使用临时文件名、调整请求间隔,问题仍未解决,相关代码如下:
let mediaRecorder; let audioChunks = []; let sendInterval; document.getElementById('recordBtn').addEventListener('click', async () => { if (mediaRecorder && mediaRecorder.state === "recording") { mediaRecorder.stop(); clearInterval(sendInterval); // Clear the interval when stopping document.getElementById('recordBtn').textContent = 'Start Recording'; } else { try { const stream = await navigator.mediaDevices.getUserMedia({ audio: true }); mediaRecorder = new MediaRecorder(stream); mediaRecorder.ondataavailable = event => { if (event.data.size > 0) { audioChunks.push(event.data); console.log('Chunk received:', event.data.size); } }; // Start recording with timeslice to ensure chunks are generated at regular intervals mediaRecorder.start(3000); document.getElementById('recordBtn').textContent = 'Stop Recording'; var x = 1; const sendAudioChunks = async () => { if (audioChunks.length > 0) { console.log('Sending chunks:', audioChunks.length); const audioBlob = new Blob(audioChunks, { 'type': 'audio/wav' }); audioChunks = []; // Clear chunks after sending const formData = new FormData(); formData.append('audio_file', audioBlob, 'audio.wav'); try { // Inside your try block within sendAudioChunks const response = await fetch('/transcribe', { method: 'POST', body: formData, }); if (!response.ok) { // Log or handle non-200 responses here console.error('Server responded with non-200 status:', response.status); } const data = await response.json(); console.log('Server response:', data); document.getElementById('transcriptionContainer').textContent = data.transcription || 'Transcription failed or was empty.'; } catch (error) { console.error('Failed to send audio chunks:', error); } } else { console.log('No chunks to send'); } }; clearInterval(sendInterval); sendInterval = setInterval(sendAudioChunks, 3000); mediaRecorder.onstop = async () => { clearInterval(sendInterval); // Clear the interval when stopping await sendAudioChunks(); // Send any remaining chunks }; } catch (error) { console.error('Error accessing media devices:', error); } } });
问题根源
核心错误是强行将MediaRecorder生成的音频片段标记为WAV格式,但这些片段本身并不是完整的WAV文件。
MediaRecorder默认输出WebM或OGG格式(取决于浏览器),用start(3000)生成的chunk是这些格式的片段,缺少WAV文件必需的文件头信息。首次请求可能因为API容错性能被识别,但后续片段只有纯音频数据,没有格式标识头,导致API无法解码,返回400错误。
解决方案
有两种可行的修复方向:
方向1:使用MediaRecorder原生格式发送(推荐,无需转换)
如果API支持WebM/OGG格式,直接用MediaRecorder的实际输出格式发送即可:
let mediaRecorder; let audioChunks = []; let sendInterval; document.getElementById('recordBtn').addEventListener('click', async () => { if (mediaRecorder && mediaRecorder.state === "recording") { mediaRecorder.stop(); clearInterval(sendInterval); document.getElementById('recordBtn').textContent = 'Start Recording'; } else { try { const stream = await navigator.mediaDevices.getUserMedia({ audio: true }); mediaRecorder = new MediaRecorder(stream); // 获取MediaRecorder实际使用的MIME类型 const mimeType = mediaRecorder.mimeType || 'audio/webm'; console.log('当前录音格式:', mimeType); mediaRecorder.ondataavailable = event => { if (event.data.size > 0) { audioChunks.push(event.data); console.log('收到音频片段:', event.data.size); } }; mediaRecorder.start(3000); document.getElementById('recordBtn').textContent = '停止录音'; const sendAudioChunks = async () => { if (audioChunks.length > 0) { console.log('发送片段数量:', audioChunks.length); // 使用原生格式创建Blob const audioBlob = new Blob(audioChunks, { 'type': mimeType }); audioChunks = []; const formData = new FormData(); // 文件名后缀和格式匹配 const fileName = `audio.${mimeType.split('/')[1]}`; formData.append('audio_file', audioBlob, fileName); try { const response = await fetch('/transcribe', { method: 'POST', body: formData, }); if (!response.ok) { const errorDetails = await response.text(); console.error('服务器错误:', response.status, errorDetails); return; } const data = await response.json(); console.log('转写结果:', data); document.getElementById('transcriptionContainer').textContent = data.transcription || '无有效转写结果'; } catch (error) { console.error('发送音频失败:', error); } } else { console.log('无音频片段可发送'); } }; clearInterval(sendInterval); sendInterval = setInterval(sendAudioChunks, 3000); mediaRecorder.onstop = async () => { clearInterval(sendInterval); await sendAudioChunks(); }; } catch (error) { console.error('访问麦克风失败:', error); } } });
方向2:将片段转换为完整WAV格式(若API仅支持WAV)
如果API只接受WAV格式,用Web Audio API将MediaRecorder输出转换为带完整头的WAV文件:
首先添加转换函数:
async function blobToWav(blob) { const arrayBuffer = await blob.arrayBuffer(); const audioContext = new (window.AudioContext || window.webkitAudioContext)(); const audioBuffer = await audioContext.decodeAudioData(arrayBuffer); // 创建WAV文件头和数据 const length = audioBuffer.length * audioBuffer.numberOfChannels * 2 + 44; const arrayBufferWav = new ArrayBuffer(length); const view = new DataView(arrayBufferWav); const channels = []; let offset = 0; let pos = 0; // 写入WAV头信息 const setUint16 = (data) => { view.setUint16(pos, data, true); pos += 2; }; const setUint32 = (data) => { view.setUint32(pos, data, true); pos += 4; }; setUint32(0x46464952); // "RIFF" setUint32(length - 8); // 文件总长度-8 setUint32(0x45564157); // "WAVE" setUint32(0x20746d66); // "fmt " setUint32(16); // PCM格式块大小 setUint16(1); // PCM编码类型 setUint16(audioBuffer.numberOfChannels); // 声道数 setUint32(audioBuffer.sampleRate); // 采样率 setUint32(audioBuffer.sampleRate * 2 * audioBuffer.numberOfChannels); // 字节率 setUint16(audioBuffer.numberOfChannels * 2); // 块对齐 setUint16(16); // 位深度 setUint32(0x61746164); // "data" setUint32(length - pos - 4); // 音频数据长度 // 写入音频采样数据 for (let i = 0; i < audioBuffer.numberOfChannels; i++) { channels.push(audioBuffer.getChannelData(i)); } while (pos < length) { for (let i = 0; i < audioBuffer.numberOfChannels; i++) { let sample = Math.max(-1, Math.min(1, channels[i][offset])); sample = sample < 0 ? sample * 0x8000 : sample * 0x7FFF; view.setInt16(pos, sample, true); pos += 2; } offset++; } return new Blob([arrayBufferWav], { type: 'audio/wav' }); }
然后在sendAudioChunks中调用转换:
const sendAudioChunks = async () => { if (audioChunks.length > 0) { console.log('发送片段数量:', audioChunks.length); const audioBlob = new Blob(audioChunks, { 'type': mediaRecorder.mimeType }); audioChunks = []; // 转换为WAV格式 const wavBlob = await blobToWav(audioBlob); const formData = new FormData(); formData.append('audio_file', wavBlob, 'audio.wav'); try { const response = await fetch('/transcribe', { method: 'POST', body: formData, }); if (!response.ok) { const errorDetails = await response.text(); console.error('服务器错误:', response.status, errorDetails); return; } const data = await response.json(); console.log('转写结果:', data); document.getElementById('transcriptionContainer').textContent = data.transcription || '无有效转写结果'; } catch (error) { console.error('发送音频失败:', error); } } else { console.log('无音频片段可发送'); } };
额外注意事项
- 测试时可在控制台打印
mediaRecorder.mimeType,确认当前录音格式 - WAV转换会有一定性能开销,短片段影响不大,长片段建议优化
- 若API支持流式转写,可考虑用WebSocket代替定时POST,效率更高
内容的提问来源于stack exchange,提问作者friso
相关产品推荐
相关产品推荐

