You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

FastAPI后端WebSocket音频播放卡顿问题排查求助

WebSocket播放音频卡顿问题排查与解决

问题场景

我有一个基于FastAPI后端和Web前端的应用,想通过WebSocket播放音频(需要单一端点管理多项WebSocket交互状态),但WebSocket播放的音频非常卡顿,而使用流式GET请求播放却完全正常,二者读取文件的逻辑基本一致。

相关代码

后端代码

from fastapi import WebSocket, APIRouter
from fastapi.responses import StreamingResponse
import wave
import asyncio

router = APIRouter()

@router.websocket('/audio_ws')
async def audio_sockets(ws: WebSocket):
    await ws.accept()

    file = wave.open('my_file.wav', 'rb')
    CHUNK = 1024

    with open('paper_to_audio/data/paper.wav', 'rb') as file_like:
        while True:
            next = file_like.read(CHUNK)
            if next == b'':
                file_like.close()
                break
            await ws.send_bytes(next)


@router.get("/audio")
def read_audio():
    def iterfile():
        CHUNK = 1024
        with open('my_file.wav', 'rb') as file_like:
            while True:
                next = file_like.read(1024)
                if next == b'':
                    file_like.close()
                    break
                yield next
    return StreamingResponse(iterfile(), media_type="audio/wav")

正常工作的流式播放前端代码

<audio preload="none" controls id="audio">
   <source src="/audio" type="audio/wav">
</audio>

WebSocket播放前端代码

function playAudioFromBackend() {
    const sample_rate = 44100; // Hz

    // Websocket url
    const ws_url = "ws://localhost:8000/audio_ws"

    let audio_context = null;
    let ws = null;

    async function start() {
        if (ws != null) {
            return;
        }

        // Create an AudioContext that plays audio from the AudioWorkletNode  
        audio_context = new AudioContext();
        await audio_context.audioWorklet.addModule('audioProcessor.js');
        const audioNode = new AudioWorkletNode(audio_context, 'audio-processor');
        audioNode.connect(audio_context.destination);

        // Setup the websocket 
        ws = new WebSocket(ws_url);
        ws.binaryType = 'arraybuffer';

        // Process incoming messages
        ws.onmessage = (event) => {
            // Convert to Float32 lpcm, which is what AudioWorkletNode expects
            const int16Array = new Int16Array(event.data);
            let float32Array = new Float32Array(int16Array.length);
            for (let i = 0; i < int16Array.length; i++) {
                float32Array[i] = int16Array[i] / 32768.;
            }

            // Send the audio data to the AudioWorkletNode
            audioNode.port.postMessage({ message: 'audioData', audioData: float32Array });
        }

        ws.onopen = () => {
            console.log('WebSocket connection opened.');
        };

        ws.onclose = () => {
            console.log('WebSocket connection closed.');
        };

        ws.onerror = error => {
            console.error('WebSocket error:', error);
        };
    }

    async function stop() {
        console.log('Stopping audio');
        if (audio_context) {
            await audio_context.close();
            audio_context = null;
            ws.close();
            ws = null;
        }
    }

    start()
}

AudioWorklet代码

class AudioProcessor extends AudioWorkletProcessor {

    constructor() {
        super();
        this.buffer = new Float32Array();

        // Receive audio data from the main thread, and add it to the buffer
        this.port.onmessage = (event) => {
            let newFetchedData = new Float32Array(this.buffer.length + event.data.audioData.length);
            newFetchedData.set(this.buffer, 0);
            newFetchedData.set(event.data.audioData, this.buffer.length);
            this.buffer = newFetchedData;
        };
    }

    // Take a chunk from the buffer and send it to the output to be played
    process(inputs, outputs, parameters) {
        const output = outputs[0];
        const channel = output[0];
        const bufferLength = this.buffer.length;
        for (let i = 0; i < channel.length; i++) {
            channel[i] = (i < bufferLength) ? this.buffer[i] : 0;
        }
        this.buffer = this.buffer.slice(channel.length);
        return true;
    }
}

registerProcessor('audio-processor', AudioProcessor);

卡顿原因分析

  1. 后端发送速率不匹配:WebSocket端点读取文件后立即发送chunk,没有匹配音频的实际播放速率,导致前端缓冲区瞬间被填满,之后又快速耗尽,出现卡顿。而浏览器的<audio>标签会自动处理流式请求的速率匹配与缓冲。
  2. 前端缓冲策略不足:AudioWorklet的缓冲没有设置预加载阈值,一旦网络延迟或后端发送不及时,缓冲就会耗尽,直接输出静音,表现为卡顿。
  3. 频繁小数据传输开销:每次只发送1024字节的chunk,对应播放时长仅约10ms,频繁的postMessage会带来额外的线程通信开销,影响数据传递效率。
  4. 主线程阻塞风险:onmessage里的Int16到Float32循环转换是同步操作,可能阻塞主线程,导致音频数据无法及时传递到Worklet。

解决方案

1. 后端控制发送速率,匹配音频播放速度

根据WAV文件的参数(采样率、位深、通道数)计算每个chunk的播放时长,发送后等待对应时间,避免过快发送:

@router.websocket('/audio_ws')
async def audio_sockets(ws: WebSocket):
    await ws.accept()

    CHUNK = 1024
    # 根据实际WAV文件参数调整
    sample_rate = 44100
    bytes_per_sample = 2  # 16位
    channels = 2
    # 计算每个chunk对应的采样帧数
    frames_per_chunk = CHUNK // (bytes_per_sample * channels)
    # 计算chunk的播放时长(秒)
    chunk_duration = frames_per_chunk / sample_rate

    with open('paper_to_audio/data/paper.wav', 'rb') as file_like:
        while True:
            next_chunk = file_like.read(CHUNK)
            if not next_chunk:
                break
            await ws.send_bytes(next_chunk)
            # 等待对应时长,匹配播放速度
            await asyncio.sleep(chunk_duration)
    await ws.close()

2. 前端优化缓冲策略,减少频繁通信

  • 积累多个小chunk后再发送给Worklet,降低postMessage频率
  • 在Worklet中设置最小缓冲阈值,避免缓冲耗尽导致静音

修改WebSocket的onmessage处理:

let audioBufferAccumulator = [];
// 积累到4096个采样点再发送(约0.09秒播放时长)
const MIN_ACCUMULATE_LENGTH = 4096;

ws.onmessage = (event) => {
    const int16Array = new Int16Array(event.data);
    // 高效转换Int16到Float32
    const float32Array = new Float32Array(int16Array.length);
    int16Array.forEach((val, idx) => {
        float32Array[idx] = val / 32768.;
    });
    
    audioBufferAccumulator.push(float32Array);
    // 计算总采样点数
    const totalLength = audioBufferAccumulator.reduce((sum, arr) => sum + arr.length, 0);
    
    if (totalLength >= MIN_ACCUMULATE_LENGTH) {
        // 合并所有积累的缓冲
        const mergedBuffer = new Float32Array(totalLength);
        let offset = 0;
        audioBufferAccumulator.forEach(arr => {
            mergedBuffer.set(arr, offset);
            offset += arr.length;
        });
        // 发送给Worklet
        audioNode.port.postMessage({ message: 'audioData', audioData: mergedBuffer });
        audioBufferAccumulator = [];
    }
}

修改AudioWorklet代码:

class AudioProcessor extends AudioWorkletProcessor {
    constructor() {
        super();
        this.buffer = new Float32Array();
        // 设置最小缓冲阈值(约0.18秒播放时长)
        this.MIN_BUFFER_THRESHOLD = 8192;

        this.port.onmessage = (event) => {
            const newData = event.data.audioData;
            const newBuffer = new Float32Array(this.buffer.length + newData.length);
            newBuffer.set(this.buffer, 0);
            newBuffer.set(newData, this.buffer.length);
            this.buffer = newBuffer;
        };
    }

    process(inputs, outputs, parameters) {
        const output = outputs[0];
        const channel = output[0];
        const neededSamples = channel.length;

        // 缓冲不足时输出静音,避免卡顿
        if (this.buffer.length < this.MIN_BUFFER_THRESHOLD) {
            channel.fill(0);
            return true;
        }

        // 取出需要的采样点
        const takeSamples = Math.min(neededSamples, this.buffer.length);
        channel.set(this.buffer.subarray(0, takeSamples));
        // 剩余位置填充静音
        if (takeSamples < neededSamples) {
            channel.fill(0, takeSamples);
        }
        // 更新缓冲
        this.buffer = this.buffer.subarray(takeSamples);
        return true;
    }
}

registerProcessor('audio-processor', AudioProcessor);

3. 优化数据转换性能

将Int16到Float32的转换改为更高效的方式,减少主线程阻塞:

const float32Array = new Float32Array(int16Array.map(val => val / 32768));

内容的提问来源于stack exchange,提问作者Nimrod Sadeh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.25 02:59:51