You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何通过WebSocket实现JS前端到Python后端的麦克风音频流式传输与处理

实时音频VAD/STT处理:修复MediaRecorder分段WebM元数据问题

我正在开发一个项目,需要在服务器端对客户端传入的音频执行实时VAD和STT处理,并且最好能将数据转换为pydub的AudioSegment以方便使用。

现有实现方案

客户端代码(JavaScript)

const socket = new WebSocket("ws://localhost:8000/audio")
const btn = document.getElementById("mic")
let stream, recorder, blobbed = null
const constraints = { audio: {
    sampleRate: 16000,
    channelCount: 1,
    volume: 1.0,
    echoCancellation: true,
    noiseSuppression: true,
    autoGainControl: true
}}
const options = {
    mimeType: "audio/webm;codecs=opus",
}
const timeSliceMs = 3 * 1000 
let toggle = false

function start_recorder() {
    recorder.start(timeSliceMs)
}

function init_recorder() {
    navigator.mediaDevices.getUserMedia(constraints)
        .then(stream => {
            recorder = new MediaRecorder(stream, options)
            recorder.ondataavailable = e => {
                if (toggle) {
                    blobbed = new Blob([e.data], { "type": options.mimeType })
                    socket.send(blobbed)
                }
            }        
            recorder.onstart = e => {
                btn.innerText = "ON"
            }
            recorder.onstop = e => {
                btn.innerText = "OFF"
            }

            start_recorder()
        })
        .catch(() => { console.error("Couldn't get user media!") })
}

function btnSwitch(e) {
    if (recorder == null) { 
        toggle = true
        init_recorder()
    } else {
        toggle = !toggle
        if (toggle) {
            start_recorder()
        } else {
            recorder.stop()
        }
    }
}

btn.addEventListener("click", btnSwitch)

socket.addEventListener("message", e => {
    let data = e.data
    try {
        data = JSON.stringify()
    } catch (err) { 
        console.error("Server error!") 
        return
    }
    console.log(data)
})

function close() {
    toggle = false
    recorder.stop()
    recorder = null
    btn.removeEventListener("click", btnSwitch)
    console.log("WS connection closed.")
}

socket.addEventListener("close", close)
socket.addEventListener("error", close)

服务器代码(Python/FastAPI)

from fastapi import FastAPI, WebSocket
from fastapi.responses import HTMLResponse
from fastapi.staticfiles import StaticFiles
from pydub import AudioSegment
from webrtcvad import Vad
import speech_recognition as sr
import numpy as np
from io import BytesIO

app = FastAPI()
app.mount('/static', StaticFiles(directory='static', html=True), name='static')


@app.get("/audio")
def audio():
    with open("audio.html", "r") as f:
        html = f.read()
    return HTMLResponse(html)

@app.websocket("/audio")
async def audio(connection: WebSocket):
    await connection.accept()
    pad = await connection.receive_bytes()
    audio_segment = AudioSegment.from_file(BytesIO(pad), codec="opus", format="webm")
    pad_len = len(audio_segment)
    
    while True:
        data = await connection.receive_bytes()
        data = pad + data
        audio_segment = AudioSegment.from_file(BytesIO(data), codec="opus", format="webm")
        
        # Preforming VAD STT

当前存在的问题

JS的MediaRecorder原本设计为在stop事件时发送完整的大Blob,仅第一批数据包含必要的元数据,因此后端需要用第一批数据填充后续每个段后再对半切片,这种方式并不理想,希望找到更合适的实现方法。

解决方案

方案一:直接发送原始PCM数据(推荐)

放弃MediaRecorder的分段Blob,直接获取音频流的原始PCM数据,每个数据包独立可解析,无需依赖初始元数据:

客户端修改代码

const socket = new WebSocket("ws://localhost:8000/audio")
const btn = document.getElementById("mic")
let stream, audioContext, scriptProcessor, toggle = false
const constraints = { audio: {
    sampleRate: 16000,
    channelCount: 1,
    volume: 1.0,
    echoCancellation: true,
    noiseSuppression: true,
    autoGainControl: true
}}

function initAudioProcessing() {
    audioContext = new AudioContext({ sampleRate: 16000 })
    const source = audioContext.createMediaStreamSource(stream)
    // 创建脚本处理器,每次处理4096帧(对应256ms@16kHz)
    scriptProcessor = audioContext.createScriptProcessor(4096, 1, 1)
    
    scriptProcessor.onaudioprocess = function(e) {
        if (!toggle) return
        const inputBuffer = e.inputBuffer
        const channelData = inputBuffer.getChannelData(0)
        
        // 将Float32Array转换为16位IntArray(WebRTC VAD需要的格式)
        const int16Array = new Int16Array(channelData.length)
        for (let i = 0; i < channelData.length; i++) {
            const sample = Math.max(-1, Math.min(1, channelData[i]))
            int16Array[i] = sample < 0 ? sample * 0x8000 : sample * 0x7FFF
        }
        
        // 通过WebSocket发送二进制数据
        socket.send(int16Array.buffer)
    }
    
    source.connect(scriptProcessor)
    scriptProcessor.connect(audioContext.destination)
}

function btnSwitch(e) {
    if (!stream) {
        toggle = true
        navigator.mediaDevices.getUserMedia(constraints)
            .then(mediaStream => {
                stream = mediaStream
                initAudioProcessing()
                btn.innerText = "ON"
            })
            .catch(() => { console.error("Couldn't get user media!") })
    } else {
        toggle = !toggle
        btn.innerText = toggle ? "ON" : "OFF"
    }
}

btn.addEventListener("click", btnSwitch)

socket.addEventListener("message", e => {
    console.log(e.data)
})

function close() {
    toggle = false
    if (scriptProcessor) scriptProcessor.disconnect()
    if (audioContext) audioContext.close()
    if (stream) stream.getTracks().forEach(track => track.stop())
    btn.removeEventListener("click", btnSwitch)
    console.log("WS connection closed.")
}

socket.addEventListener("close", close)
socket.addEventListener("error", close)

服务器修改代码

from fastapi import FastAPI, WebSocket
from fastapi.responses import HTMLResponse
from fastapi.staticfiles import StaticFiles
from pydub import AudioSegment
from webrtcvad import Vad
import numpy as np
from io import BytesIO

app = FastAPI()
app.mount('/static', StaticFiles(directory='static', html=True), name='static')

# 初始化VAD(模式3为最严格)
vad = Vad(3)

@app.get("/audio")
def audio():
    with open("audio.html", "r") as f:
        html = f.read()
    return HTMLResponse(html)

@app.websocket("/audio")
async def audio(connection: WebSocket):
    await connection.accept()
    # 音频参数:16kHz采样率,16位单声道
    sample_rate = 16000
    sample_width = 2
    channels = 1
    
    while True:
        try:
            # 接收PCM二进制数据
            pcm_data = await connection.receive_bytes()
            # 转换为numpy数组
            int16_array = np.frombuffer(pcm_data, dtype=np.int16)
            
            # 实时VAD处理(按30ms切片,WebRTC VAD支持10/20/30ms帧)
            frame_duration_ms = 30
            frame_size = int(sample_rate * frame_duration_ms / 1000)
            for i in range(0, len(int16_array), frame_size):
                frame = int16_array[i:i+frame_size]
                if len(frame) < frame_size:
                    break
                is_speech = vad.is_speech(frame.tobytes(), sample_rate)
                if is_speech:
                    print("检测到语音")
            
            # 将PCM转换为AudioSegment
            audio_segment = AudioSegment(
                data=int16_array.tobytes(),
                sample_width=sample_width,
                frame_rate=sample_rate,
                channels=channels
            )
            
            # 这里可以继续进行STT处理
            # ...
            
        except Exception as e:
            print(f"连接错误: {e}")
            break

方案二:维护连续WebM流(兼容原有编码)

如果坚持使用WebM/Opus格式,可在客户端维护完整Blob缓存,仅发送新增片段,服务器端维护连续流:

客户端修改代码

const socket = new WebSocket("ws://localhost:8000/audio")
const btn = document.getElementById("mic")
let stream, recorder, fullBlob = new Blob(), toggle = false
const constraints = { audio: {
    sampleRate: 16000,
    channelCount: 1,
    volume: 1.0,
    echoCancellation: true,
    noiseSuppression: true,
    autoGainControl: true
}}
const options = {
    mimeType: "audio/webm;codecs=opus",
}
const timeSliceMs = 3 * 1000 

function start_recorder() {
    recorder.start(timeSliceMs)
}

function init_recorder() {
    navigator.mediaDevices.getUserMedia(constraints)
        .then(stream => {
            recorder = new MediaRecorder(stream, options)
            recorder.ondataavailable = e => {
                if (!toggle) return
                // 将新数据添加到完整Blob
                const prevSize = fullBlob.size
                fullBlob = new Blob([fullBlob, e.data], { type: options.mimeType })
                // 只发送新增的部分
                const newData = fullBlob.slice(prevSize)
                socket.send(newData)
            }        
            recorder.onstart = e => {
                btn.innerText = "ON"
            }
            recorder.onstop = e => {
                btn.innerText = "OFF"
                fullBlob = new Blob()
            }

            start_recorder()
        })
        .catch(() => { console.error("Couldn't get user media!") })
}

function btnSwitch(e) {
    if (recorder == null) { 
        toggle = true
        init_recorder()
    } else {
        toggle = !toggle
        if (toggle) {
            start_recorder()
        } else {
            recorder.stop()
        }
    }
}

btn.addEventListener("click", btnSwitch)

socket.addEventListener("message", e => {
    console.log(e.data)
})

function close() {
    toggle = false
    if (recorder) recorder.stop()
    recorder = null
    btn.removeEventListener("click", btnSwitch)
    console.log("WS connection closed.")
}

socket.addEventListener("close", close)
socket.addEventListener("error", close)

服务器修改代码

from fastapi import FastAPI, WebSocket
from fastapi.responses import HTMLResponse
from fastapi.staticfiles import StaticFiles
from pydub import AudioSegment
from webrtcvad import Vad
import numpy as np
from io import BytesIO

app = FastAPI()
app.mount('/static', StaticFiles(directory='static', html=True), name='static')

@app.get("/audio")
def audio():
    with open("audio.html", "r") as f:
        html = f.read()
    return HTMLResponse(html)

@app.websocket("/audio")
async def audio(connection: WebSocket):
    await connection.accept()
    # 维护完整的WebM流缓存
    webm_buffer = BytesIO()
    
    while True:
        try:
            data = await connection.receive_bytes()
            # 将新数据写入缓存
            webm_buffer.write(data)
            # 将缓存指针移到开头
            webm_buffer.seek(0)
            # 解析完整的WebM流
            audio_segment = AudioSegment.from_file(webm_buffer, codec="opus", format="webm")
            
            # 处理最新的3秒片段(匹配客户端timeSliceMs)
            recent_segment = audio_segment[-3000:]
            # 这里进行VAD/STT处理
            # ...
            
        except Exception as e:
            print(f"连接错误: {e}")
            break

内容的提问来源于stack exchange,提问作者Peter Jurák

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.16 21:59:53