You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于SignalR、.NET和JS的实时音频流转Google语音识别返回空白问题

实时语音识别无结果问题排查与修复

核心问题分析

你遇到的空白识别结果,主要由音频格式不匹配、流连接管理错误、结果读取逻辑缺陷这三个关键问题导致:

1. 音频编码与Google配置不兼容

前端生成的是16位PCM格式的音频数据,但后端Google Speech-to-Text配置的编码是WebmOpus,两者完全不匹配,导致Google无法解析音频内容。

2. 实时流连接未持续维护

当前后端每次收到音频片段都新建一个StreamingRecognize调用,这不符合实时流的工作模式——Google需要一个持续的连接来接收连续的音频流,而非独立的片段请求。

3. 识别结果读取逻辑错误

后端代码仅尝试读取一次resultsAsyncResponseStream.Current,但GetResponseStream是异步数据流,需要循环监听直到流结束,否则会完全错过识别结果。

4. 采样率未强制对齐

前端AudioContext默认采样率可能不等于Google配置的16000Hz,采样率不匹配会导致音频失真,识别失败。


修正后的代码实现

前端JavaScript代码调整

var connection = new signalR.HubConnectionBuilder()
    .withUrl("/speechHub")
    .configureLogging(signalR.LogLevel.Information)
    .build();

async function webaudio_tooling_obj() {
    // 强制使用16000Hz采样率,匹配Google配置
    var audioContext = new AudioContext({ sampleRate: 16000 });
    console.log("audio is starting up ...");

    var buffSize = 16384;
    var microphoneStream = null;
    var scriptProcessorNode = null;

    if (!navigator.mediaDevices.getUserMedia) {
        alert('getUserMedia not supported in this browser.');
        return;
    }

    try {
        const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
        startMicrophone(stream);
    } catch (e) {
        alert('Error capturing audio.');
    }

    await connection.start().catch(err => console.error(err.toString()));

    // 转换Float32Array为16位PCM(Linear16)
    function float32ToInt16(float32Array) {
        const int16Array = new Int16Array(float32Array.length);
        for (let i = 0; i < float32Array.length; i++) {
            // 归一化到Int16范围
            const clamped = Math.max(-1, Math.min(1, float32Array[i]));
            int16Array[i] = clamped < 0 ? clamped * 0x8000 : clamped * 0x7FFF;
        }
        return int16Array;
    }

    function process_microphone_buffer(event) {
        const microphoneBuffer = event.inputBuffer.getChannelData(0);
        const int16Buffer = float32ToInt16(microphoneBuffer);
        
        // 直接转换为Base64,无需Blob中转
        const base64 = btoa(String.fromCharCode.apply(null, int16Buffer));
        sendBase64ToServer(base64);
    }

    function sendBase64ToServer(base64String) {
        connection.invoke("SendMicrophoneBuffer", base64String)
            .catch(err => console.error(err));
    }

    connection.on("ReceiveRecognizeResult", function (result) {
        console.log("识别结果:", result);
        // 这里可以更新UI展示实时识别结果
    });

    function startMicrophone(stream) {
        microphoneStream = audioContext.createMediaStreamSource(stream);
        
        scriptProcessorNode = audioContext.createScriptProcessor(buffSize, 1, 1);
        scriptProcessorNode.onaudioprocess = process_microphone_buffer;
        
        microphoneStream.connect(scriptProcessorNode);
        scriptProcessorNode.connect(audioContext.destination);
    }
};

后端C#代码调整

SpeechHub(维护每个连接的流会话)

public class SpeechHub : Hub
{
    private readonly IGoogleSpeechRecognitionService _speechService;
    // 为每个连接维护一个流会话
    private static readonly Dictionary<string, AsyncStreamingCall<StreamingRecognizeRequest, StreamingRecognizeResponse>> _activeStreams = new();

    public SpeechHub(IGoogleSpeechRecognitionService speechService)
    {
        _speechService = speechService;
    }

    public async Task StartStreaming()
    {
        var connectionId = Context.ConnectionId;
        if (_activeStreams.ContainsKey(connectionId)) return;

        // 创建新的流会话并启动结果监听
        var streamingCall = await _speechService.StartStreamingRecognitionAsync();
        _activeStreams[connectionId] = streamingCall;
        
        // 后台监听识别结果并推送给客户端
        _ = Task.Run(async () =>
        {
            await foreach (var response in streamingCall.ResponseStream.ReadAllAsync())
            {
                foreach (var result in response.Results)
                {
                    foreach (var alternative in result.Alternatives)
                    {
                        await Clients.Client(connectionId).SendAsync("ReceiveRecognizeResult", alternative.Transcript);
                    }
                }
            }
        });
    }

    public async Task SendMicrophoneBuffer(string base64AudioData)
    {
        var connectionId = Context.ConnectionId;
        if (!_activeStreams.TryGetValue(connectionId, out var streamingCall))
        {
            throw new InvalidOperationException("请先启动流会话");
        }

        var byteData = Convert.FromBase64String(base64AudioData);
        await streamingCall.RequestStream.WriteAsync(new StreamingRecognizeRequest
        {
            AudioContent = Google.Protobuf.ByteString.CopyFrom(byteData)
        });
    }

    public override async Task OnDisconnectedAsync(Exception exception)
    {
        var connectionId = Context.ConnectionId;
        if (_activeStreams.TryGetValue(connectionId, out var streamingCall))
        {
            await streamingCall.RequestStream.WriteCompleteAsync();
            _activeStreams.Remove(connectionId);
        }
        await base.OnDisconnectedAsync(exception);
    }
}

Google Speech服务实现

public interface IGoogleSpeechRecognitionService
{
    Task<AsyncStreamingCall<StreamingRecognizeRequest, StreamingRecognizeResponse>> StartStreamingRecognitionAsync();
}

public class GoogleSpeechRecognitionService : IGoogleSpeechRecognitionService
{
    private readonly SpeechClient _speechClient;

    public GoogleSpeechRecognitionService(SpeechClient speechClient)
    {
        _speechClient = speechClient;
    }

    public async Task<AsyncStreamingCall<StreamingRecognizeRequest, StreamingRecognizeResponse>> StartStreamingRecognitionAsync()
    {
        var streamingCall = _speechClient.StreamingRecognize();
        
        // 发送配置请求,编码改为Linear16(对应前端的16位PCM)
        await streamingCall.RequestStream.WriteAsync(new StreamingRecognizeRequest
        {
            StreamingConfig = new StreamingRecognitionConfig
            {
                Config = new RecognitionConfig
                {
                    Encoding = RecognitionConfig.Types.AudioEncoding.Linear16,
                    SampleRateHertz = 16000,
                    LanguageCode = "en-US",
                    EnableAutomaticPunctuation = true
                },
                InterimResults = true // 返回中间识别结果
            }
        });

        return streamingCall;
    }
}

关键修正说明

  1. 音频格式对齐:前端生成Linear16格式(16位PCM),后端Google配置对应编码,确保音频可被解析。
  2. 持续流会话:为每个客户端连接维护一个独立的流会话,避免每次片段都重建连接。
  3. 正确读取异步结果:使用await foreach遍历ResponseStream,确保获取所有识别结果。
  4. 强制采样率:前端AudioContext强制设置16000Hz,与Google配置一致。

内容的提问来源于stack exchange,提问作者Amit Guleria

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 17:07:34