如何通过WebSocket实现JS前端到Python后端的麦克风音频流式传输与处理
实时音频VAD/STT处理:修复MediaRecorder分段WebM元数据问题
我正在开发一个项目,需要在服务器端对客户端传入的音频执行实时VAD和STT处理,并且最好能将数据转换为pydub的AudioSegment以方便使用。
现有实现方案
客户端代码(JavaScript)
const socket = new WebSocket("ws://localhost:8000/audio") const btn = document.getElementById("mic") let stream, recorder, blobbed = null const constraints = { audio: { sampleRate: 16000, channelCount: 1, volume: 1.0, echoCancellation: true, noiseSuppression: true, autoGainControl: true }} const options = { mimeType: "audio/webm;codecs=opus", } const timeSliceMs = 3 * 1000 let toggle = false function start_recorder() { recorder.start(timeSliceMs) } function init_recorder() { navigator.mediaDevices.getUserMedia(constraints) .then(stream => { recorder = new MediaRecorder(stream, options) recorder.ondataavailable = e => { if (toggle) { blobbed = new Blob([e.data], { "type": options.mimeType }) socket.send(blobbed) } } recorder.onstart = e => { btn.innerText = "ON" } recorder.onstop = e => { btn.innerText = "OFF" } start_recorder() }) .catch(() => { console.error("Couldn't get user media!") }) } function btnSwitch(e) { if (recorder == null) { toggle = true init_recorder() } else { toggle = !toggle if (toggle) { start_recorder() } else { recorder.stop() } } } btn.addEventListener("click", btnSwitch) socket.addEventListener("message", e => { let data = e.data try { data = JSON.stringify() } catch (err) { console.error("Server error!") return } console.log(data) }) function close() { toggle = false recorder.stop() recorder = null btn.removeEventListener("click", btnSwitch) console.log("WS connection closed.") } socket.addEventListener("close", close) socket.addEventListener("error", close)
服务器代码(Python/FastAPI)
from fastapi import FastAPI, WebSocket from fastapi.responses import HTMLResponse from fastapi.staticfiles import StaticFiles from pydub import AudioSegment from webrtcvad import Vad import speech_recognition as sr import numpy as np from io import BytesIO app = FastAPI() app.mount('/static', StaticFiles(directory='static', html=True), name='static') @app.get("/audio") def audio(): with open("audio.html", "r") as f: html = f.read() return HTMLResponse(html) @app.websocket("/audio") async def audio(connection: WebSocket): await connection.accept() pad = await connection.receive_bytes() audio_segment = AudioSegment.from_file(BytesIO(pad), codec="opus", format="webm") pad_len = len(audio_segment) while True: data = await connection.receive_bytes() data = pad + data audio_segment = AudioSegment.from_file(BytesIO(data), codec="opus", format="webm") # Preforming VAD STT
当前存在的问题
JS的MediaRecorder原本设计为在stop事件时发送完整的大Blob,仅第一批数据包含必要的元数据,因此后端需要用第一批数据填充后续每个段后再对半切片,这种方式并不理想,希望找到更合适的实现方法。
解决方案
方案一:直接发送原始PCM数据(推荐)
放弃MediaRecorder的分段Blob,直接获取音频流的原始PCM数据,每个数据包独立可解析,无需依赖初始元数据:
客户端修改代码
const socket = new WebSocket("ws://localhost:8000/audio") const btn = document.getElementById("mic") let stream, audioContext, scriptProcessor, toggle = false const constraints = { audio: { sampleRate: 16000, channelCount: 1, volume: 1.0, echoCancellation: true, noiseSuppression: true, autoGainControl: true }} function initAudioProcessing() { audioContext = new AudioContext({ sampleRate: 16000 }) const source = audioContext.createMediaStreamSource(stream) // 创建脚本处理器,每次处理4096帧(对应256ms@16kHz) scriptProcessor = audioContext.createScriptProcessor(4096, 1, 1) scriptProcessor.onaudioprocess = function(e) { if (!toggle) return const inputBuffer = e.inputBuffer const channelData = inputBuffer.getChannelData(0) // 将Float32Array转换为16位IntArray(WebRTC VAD需要的格式) const int16Array = new Int16Array(channelData.length) for (let i = 0; i < channelData.length; i++) { const sample = Math.max(-1, Math.min(1, channelData[i])) int16Array[i] = sample < 0 ? sample * 0x8000 : sample * 0x7FFF } // 通过WebSocket发送二进制数据 socket.send(int16Array.buffer) } source.connect(scriptProcessor) scriptProcessor.connect(audioContext.destination) } function btnSwitch(e) { if (!stream) { toggle = true navigator.mediaDevices.getUserMedia(constraints) .then(mediaStream => { stream = mediaStream initAudioProcessing() btn.innerText = "ON" }) .catch(() => { console.error("Couldn't get user media!") }) } else { toggle = !toggle btn.innerText = toggle ? "ON" : "OFF" } } btn.addEventListener("click", btnSwitch) socket.addEventListener("message", e => { console.log(e.data) }) function close() { toggle = false if (scriptProcessor) scriptProcessor.disconnect() if (audioContext) audioContext.close() if (stream) stream.getTracks().forEach(track => track.stop()) btn.removeEventListener("click", btnSwitch) console.log("WS connection closed.") } socket.addEventListener("close", close) socket.addEventListener("error", close)
服务器修改代码
from fastapi import FastAPI, WebSocket from fastapi.responses import HTMLResponse from fastapi.staticfiles import StaticFiles from pydub import AudioSegment from webrtcvad import Vad import numpy as np from io import BytesIO app = FastAPI() app.mount('/static', StaticFiles(directory='static', html=True), name='static') # 初始化VAD(模式3为最严格) vad = Vad(3) @app.get("/audio") def audio(): with open("audio.html", "r") as f: html = f.read() return HTMLResponse(html) @app.websocket("/audio") async def audio(connection: WebSocket): await connection.accept() # 音频参数:16kHz采样率,16位单声道 sample_rate = 16000 sample_width = 2 channels = 1 while True: try: # 接收PCM二进制数据 pcm_data = await connection.receive_bytes() # 转换为numpy数组 int16_array = np.frombuffer(pcm_data, dtype=np.int16) # 实时VAD处理(按30ms切片,WebRTC VAD支持10/20/30ms帧) frame_duration_ms = 30 frame_size = int(sample_rate * frame_duration_ms / 1000) for i in range(0, len(int16_array), frame_size): frame = int16_array[i:i+frame_size] if len(frame) < frame_size: break is_speech = vad.is_speech(frame.tobytes(), sample_rate) if is_speech: print("检测到语音") # 将PCM转换为AudioSegment audio_segment = AudioSegment( data=int16_array.tobytes(), sample_width=sample_width, frame_rate=sample_rate, channels=channels ) # 这里可以继续进行STT处理 # ... except Exception as e: print(f"连接错误: {e}") break
方案二:维护连续WebM流(兼容原有编码)
如果坚持使用WebM/Opus格式,可在客户端维护完整Blob缓存,仅发送新增片段,服务器端维护连续流:
客户端修改代码
const socket = new WebSocket("ws://localhost:8000/audio") const btn = document.getElementById("mic") let stream, recorder, fullBlob = new Blob(), toggle = false const constraints = { audio: { sampleRate: 16000, channelCount: 1, volume: 1.0, echoCancellation: true, noiseSuppression: true, autoGainControl: true }} const options = { mimeType: "audio/webm;codecs=opus", } const timeSliceMs = 3 * 1000 function start_recorder() { recorder.start(timeSliceMs) } function init_recorder() { navigator.mediaDevices.getUserMedia(constraints) .then(stream => { recorder = new MediaRecorder(stream, options) recorder.ondataavailable = e => { if (!toggle) return // 将新数据添加到完整Blob const prevSize = fullBlob.size fullBlob = new Blob([fullBlob, e.data], { type: options.mimeType }) // 只发送新增的部分 const newData = fullBlob.slice(prevSize) socket.send(newData) } recorder.onstart = e => { btn.innerText = "ON" } recorder.onstop = e => { btn.innerText = "OFF" fullBlob = new Blob() } start_recorder() }) .catch(() => { console.error("Couldn't get user media!") }) } function btnSwitch(e) { if (recorder == null) { toggle = true init_recorder() } else { toggle = !toggle if (toggle) { start_recorder() } else { recorder.stop() } } } btn.addEventListener("click", btnSwitch) socket.addEventListener("message", e => { console.log(e.data) }) function close() { toggle = false if (recorder) recorder.stop() recorder = null btn.removeEventListener("click", btnSwitch) console.log("WS connection closed.") } socket.addEventListener("close", close) socket.addEventListener("error", close)
服务器修改代码
from fastapi import FastAPI, WebSocket from fastapi.responses import HTMLResponse from fastapi.staticfiles import StaticFiles from pydub import AudioSegment from webrtcvad import Vad import numpy as np from io import BytesIO app = FastAPI() app.mount('/static', StaticFiles(directory='static', html=True), name='static') @app.get("/audio") def audio(): with open("audio.html", "r") as f: html = f.read() return HTMLResponse(html) @app.websocket("/audio") async def audio(connection: WebSocket): await connection.accept() # 维护完整的WebM流缓存 webm_buffer = BytesIO() while True: try: data = await connection.receive_bytes() # 将新数据写入缓存 webm_buffer.write(data) # 将缓存指针移到开头 webm_buffer.seek(0) # 解析完整的WebM流 audio_segment = AudioSegment.from_file(webm_buffer, codec="opus", format="webm") # 处理最新的3秒片段(匹配客户端timeSliceMs) recent_segment = audio_segment[-3000:] # 这里进行VAD/STT处理 # ... except Exception as e: print(f"连接错误: {e}") break
内容的提问来源于stack exchange,提问作者Peter Jurák
相关产品推荐
相关产品推荐

