You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何基于Angular、ASP.NET Core与Azure语音服务实现实时语音转文字

实时语音转文字(Azure Speech Service + WebSocket)实现方案

问题描述

我正在尝试用Microsoft Azure Speech Service构建实时语音转文字Web应用。目前通过MediaRecorder录制用户语音,录制完成后再发送到后端转成文本,但找不到完整示例说明如何从浏览器实时采集语音并通过WebSocket发送到后端。

现有代码

现有Angular代码(录制后上传)

import { Component, OnInit, OnDestroy, NgZone } from '@angular/core';
import { DomSanitizer, SafeUrl } from '@angular/platform-browser';

import * as WavEncoder from 'wav-encoder';
import { AdminService } from '../admin.service';

@Component({
  selector: 'app-audio-recorder',
  templateUrl: './audio-recorder.component.html',
  styleUrls: ['./audio-recorder.component.css']
})
export class AudioRecorderComponent implements OnInit, OnDestroy {
  private mediaRecorder: MediaRecorder;
  private audioChunks: Blob[] = [];
  audioUrl: SafeUrl;
  isRecording = false;
  isLoading = false;
  transcript: {
    text: string,
    duration: string
  } = { text: '', duration: '' }

  constructor(
    private sanitizer: DomSanitizer,
    private adminService: AdminService,
    private ngZone: NgZone) { }

  ngOnInit(): void {
    navigator.mediaDevices.getUserMedia({ audio: true })
      .then(stream => {
        this.mediaRecorder = new MediaRecorder(stream);
        this.mediaRecorder.ondataavailable = event => {
          if (event.data.size > 0) {
            this.audioChunks.push(event.data);
          }
        };

        this.mediaRecorder.onstop = async () => {
          const audioBlob = new Blob(this.audioChunks, { type: 'audio/webm' });
          const arrayBuffer = await audioBlob.arrayBuffer();
          const audioBuffer = await this.decodeAudioData(arrayBuffer);
          const wavBlob = await this.convertToWav(audioBuffer);
          const audioURL = URL.createObjectURL(wavBlob);
          this.ngZone.run(() => {
            this.audioUrl = this.sanitizer.bypassSecurityTrustUrl(audioURL);
          });
          this.ngZone.run(() => {
            this.isLoading = true;
          });
          this.adminService.transcriptFile(wavBlob, "file")
            .subscribe({
              next: (data) => {
                this.ngZone.run(() => {
                  this.transcript.text = data
                  this.isLoading = false
                });
              },
              error: (err) => {
                console.error(err)
                this.isLoading = false
              }
            })
          this.audioChunks = [];
        };
      })
      .catch(error => {
        console.error('Error accessing microphone:', error);
      });
  }

  startRecording() {
    if (this.mediaRecorder && this.mediaRecorder.state !== 'recording') {
      this.isRecording = true
      this.mediaRecorder.start();
    } else {
      console.warn('MediaRecorder is not available or already recording');
    }
  }

  stopRecording() {
    if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
      this.isRecording = false
      this.mediaRecorder.stop();
    } else {
      console.warn('MediaRecorder is not recording');
    }
  }

  ngOnDestroy() {
    if (this.audioUrl) {
      URL.revokeObjectURL(this.audioUrl as string);
    }
  }

  // Helper function to decode audio data
  private async decodeAudioData(arrayBuffer: ArrayBuffer): Promise<AudioBuffer> {
    const audioContext = new AudioContext();
    return await audioContext.decodeAudioData(arrayBuffer);
  }

  // Helper function to convert AudioBuffer to WAV format
  private async convertToWav(audioBuffer: AudioBuffer): Promise<Blob> {
    const wavData = await WavEncoder.encode({
      sampleRate: 16000, // 16 kHz
      bitDepth: 16, // 16 bits
      channelData: [
        this.downsampleBuffer(audioBuffer.getChannelData(0), audioBuffer.sampleRate, 16000)
      ]
    });
    return new Blob([wavData], { type: 'audio/wav' });
  }

  // Helper function to downsample audio buffer
  private downsampleBuffer(buffer: Float32Array, sampleRate: number, targetRate: number): Float32Array {
    if (sampleRate === targetRate) {
      return buffer;
    }
    const sampleRateRatio = sampleRate / targetRate;
    const newLength = Math.round(buffer.length / sampleRateRatio);
    const result = new Float32Array(newLength);
    let offsetResult = 0;
    let offsetBuffer = 0;
    while (offsetResult < result.length) {
      const nextOffsetBuffer = Math.round((offsetResult + 1) * sampleRateRatio);
      let accum = 0, count = 0;
      for (let i = offsetBuffer; i < nextOffsetBuffer && i < buffer.length; i++) {
        accum += buffer[i];
        count++;
      }
      result[offsetResult] = accum / count;
      offsetResult++;
      offsetBuffer = nextOffsetBuffer;
    }
    return result;
  }
}

现有C#后端代码(批量转写)

[Route("api/[controller]")]
[ApiController]
public class AudioController : ControllerBase
{
    [HttpPost("Transcript")]
    public async Task<string> Transcript(IFormFile audioFile)
    {
        ArgumentNullException.ThrowIfNull(audioFile);
        using var audioStream = audioFile.OpenReadStream();
        Console.OutputEncoding = Encoding.UTF8;
        var audioKey = "...";
        var audioRegion = "...";
        using var fileStream = audioFile.OpenReadStream();
        var speechConfig = SpeechConfig.FromSubscription(audioKey, audioRegion);
        byte[] audioData;
        using (var memoryStream = new MemoryStream())
        {
            await audioFile.CopyToAsync(memoryStream);
            audioData = memoryStream.ToArray();
        }
        byte channels = 1;
        byte bitsPerSample = 16;
        uint samplesPerSecond = 16000; // or 8000 based on your audio file's sample rate
        var audioFormat = AudioStreamFormat.GetWaveFormatPCM(samplesPerSecond, bitsPerSample, channels);
        var audioConfig = AudioConfig.FromStreamInput(new BytesAudioStream(audioData), audioFormat);
        speechConfig.SpeechRecognitionLanguage = "ar-SA";
        var autoDetectSourceLanguageConfig =
                AutoDetectSourceLanguageConfig.FromLanguages(["ar-SA"]);
        var speechRecognizer = new SpeechRecognizer(speechConfig, autoDetectSourceLanguageConfig, audioConfig);
        Stopwatch stopwatch = Stopwatch.StartNew();
        var speechRecognitionResult = await speechRecognizer.RecognizeOnceAsync();
        stopwatch.Stop();
        var duration = stopwatch.Elapsed;
        return speechRecognitionResult.Text;
    }
}

public class BytesAudioStream(byte[] audioData) : PullAudioInputStreamCallback
{
    private readonly MemoryStream memoryStream = new(audioData);
    public override int Read(byte[] buffer, uint size)
    {
        return memoryStream.Read(buffer, 0, (int)size);
    }
    public override void Close()
    {
        memoryStream.Close();
    }
}

实时WebSocket传输实现方案

一、前端Angular修改:实时采集并发送音频数据

替换原有录制后上传的逻辑,使用AudioContext直接获取原始音频流,转换为Azure要求的16kHz、16位单声道PCM格式,通过WebSocket实时推送:

import { Component, OnInit, OnDestroy } from '@angular/core';

@Component({
  selector: 'app-realtime-audio-recorder',
  templateUrl: './realtime-audio-recorder.component.html',
  styleUrls: ['./realtime-audio-recorder.component.css']
})
export class RealtimeAudioRecorderComponent implements OnInit, OnDestroy {
  private audioContext: AudioContext;
  private mediaStream: MediaStream;
  private scriptProcessor: ScriptProcessorNode;
  private ws: WebSocket;
  isRecording = false;
  transcript = '';

  ngOnInit(): void {
    // 初始化WebSocket连接
    this.ws = new WebSocket('ws://localhost:5000/api/audio/realtime-transcript');
    this.ws.onmessage = (event) => {
      // 接收后端返回的实时识别结果
      this.transcript = event.data;
    };
    this.ws.onerror = (error) => {
      console.error('WebSocket错误:', error);
    };
    this.ws.onclose = () => {
      console.log('WebSocket连接关闭');
      this.stopRecording();
    };
  }

  async startRecording() {
    if (this.isRecording) return;
    this.isRecording = true;
    this.transcript = '';

    // 获取麦克风流,强制指定16kHz采样率、单声道
    this.mediaStream = await navigator.mediaDevices.getUserMedia({ 
      audio: { sampleRate: 16000, channelCount: 1, echoCancellation: true } 
    });
    this.audioContext = new AudioContext({ sampleRate: 16000 });
    const source = this.audioContext.createMediaStreamSource(this.mediaStream);
    
    // 创建脚本处理器,每次处理4096帧音频数据
    this.scriptProcessor = this.audioContext.createScriptProcessor(4096, 1, 1);
    source.connect(this.scriptProcessor);
    this.scriptProcessor.connect(this.audioContext.destination);

    // 实时处理并发送音频数据
    this.scriptProcessor.onaudioprocess = (event) => {
      if (!this.isRecording || this.ws.readyState !== WebSocket.OPEN) return;
      
      // 获取左声道的Float32格式音频数据
      const inputBuffer = event.inputBuffer.getChannelData(0);
      // 转换为Azure要求的16位PCM格式(Uint8Array)
      const pcmData = this.float32ToInt16(inputBuffer);
      // 通过WebSocket发送二进制数据
      this.ws.send(pcmData);
    };
  }

  stopRecording() {
    this.isRecording = false;
    // 关闭音频流和上下文
    this.mediaStream?.getTracks().forEach(track => track.stop());
    this.audioContext?.close();
    this.scriptProcessor?.disconnect();
    // 关闭WebSocket连接
    if (this.ws.readyState === WebSocket.OPEN) {
      this.ws.close();
    }
  }

  ngOnDestroy() {
    this.stopRecording();
  }

  // 将Float32音频数据转换为16位PCM格式(小端序)
  private float32ToInt16(buffer: Float32Array): Uint8Array {
    const length = buffer.length;
    const result = new Uint8Array(length * 2);
    let index = 0;
    for (let i = 0; i < length; i++) {
      let sample = buffer[i];
      // 限制音频范围在[-1, 1]
      sample = Math.max(-1, Math.min(1, sample));
      // 转换为16位整数(范围:-32768 到 32767)
      sample = sample < 0 ? sample * 32768 : sample * 32767;
      // 写入Uint8Array(小端序存储)
      result[index++] = sample & 0xff;
      result[index++] = (sample >> 8) & 0xff;
    }
    return result;
  }
}

二、后端C#修改:WebSocket服务器 + Azure实时语音识别

在ASP.NET Core中启用WebSocket中间件,接收前端推送的音频数据,实时转发至Azure Speech Service进行识别,并将结果回传前端:

1. 启用WebSocket中间件(Program.cs)

var builder = WebApplication.CreateBuilder(args);

builder.Services.AddControllers();
builder.Services.AddEndpointsApiExplorer();
builder.Services.AddSwaggerGen();

var app = builder.Build();

// 启用WebSocket中间件
app.UseWebSockets();

if (app.Environment.IsDevelopment()) {
    app.UseSwagger();
    app.UseSwaggerUI();
}

app.UseHttpsRedirection();
app.UseAuthorization();
app.MapControllers();

app.Run();

2. 实现WebSocket实时识别接口(AudioController)

using Microsoft.AspNetCore.Mvc;
using Microsoft.CognitiveServices.Speech;
using Microsoft.CognitiveServices.Speech.Audio;
using System.Buffers;
using System.Net.WebSockets;
using System.Text;

[Route("api/[controller]")]
[ApiController]
public class AudioController : ControllerBase
{
    private readonly string _speechKey = "你的Azure Speech密钥";
    private readonly string _speechRegion = "你的Azure区域(如eastus)";

    [HttpGet("realtime-transcript")]
    public async Task RealtimeTranscript()
    {
        if (!HttpContext.WebSockets.IsWebSocketRequest)
        {
            HttpContext.Response.StatusCode = StatusCodes.Status400BadRequest;
            return;
        }

        using var webSocket = await HttpContext.WebSockets.AcceptWebSocketAsync();
        // 复用内存缓冲区,减少GC开销
        var buffer = ArrayPool<byte>.Shared.Rent(4096 * 2);

        try
        {
            // 配置Azure Speech服务
            var speechConfig = SpeechConfig.FromSubscription(_speechKey, _speechRegion);
            speechConfig.SpeechRecognitionLanguage = "ar-SA";
            
            // 创建Push类型音频输入流,用于实时接收音频数据
            using var audioInputStream = AudioInputStream.CreatePushStream();
            using var audioConfig = AudioConfig.FromStreamInput(audioInputStream);
            
            // 创建实时语音识别器
            using var speechRecognizer = new SpeechRecognizer(speechConfig, audioConfig);

            // 订阅实时识别中间结果事件
            speechRecognizer.Recognizing += async (s, e) =>
            {
                if (e.Result.Reason == ResultReason.RecognizingSpeech)
                {
                    var message = Encoding.UTF8.GetBytes(e.Result.Text);
                    await webSocket.SendAsync(
                        new ArraySegment<byte>(message, 0, message.Length), 
                        WebSocketMessageType.Text, 
                        true, 
                        CancellationToken.None
                    );
                }
            };

            // 订阅最终识别结果事件
            speechRecognizer.Recognized += async (s, e) =>
            {
                if (e.Result.Reason == ResultReason.RecognizedSpeech)
                {
                    var message = Encoding.UTF8.GetBytes(e.Result.Text);
                    await webSocket.SendAsync(
                        new ArraySegment<byte>(message, 0, message.Length), 
                        WebSocketMessageType.Text, 
                        true, 
                        CancellationToken.None
                    );
                }
                else if (e.Result.Reason == ResultReason.NoMatch)
                {
                    var message = Encoding.UTF8.GetBytes("未识别到有效语音");
                    await webSocket.SendAsync(
                        new ArraySegment<byte>(message, 0, message.Length), 
                        WebSocketMessageType.Text, 
                        true, 
                        CancellationToken.None
                    );
                }
            };

            // 启动连续识别
            await speechRecognizer.StartContinuousRecognitionAsync();

            // 循环接收WebSocket的音频数据
            WebSocketReceiveResult result;
            do
            {
                result = await webSocket.ReceiveAsync(new ArraySegment<byte>(buffer), CancellationToken.None);
                if (result.MessageType == WebSocketMessageType.Binary)
                {
                    // 将接收到的PCM数据推送到Azure Speech流
                    audioInputStream.Write(buffer, 0, result.Count);
                }
            } while (!result.CloseStatus.HasValue);

            // 停止识别
            await speechRecognizer.StopContinuousRecognitionAsync();
        }
        catch (Exception ex)
        {
            Console.Error.WriteLine($"实时识别错误:{ex.Message}");
        }
        finally
        {
            // 归还内存缓冲区
            ArrayPool<byte>.Shared.Return(buffer);
            await webSocket.CloseAsync(WebSocketCloseStatus.NormalClosure, "连接正常关闭", CancellationToken.None);
        }
    }

    // 保留原有批量转写接口(可选)
    [HttpPost("Transcript")]
    public async Task<string> Transcript(IFormFile audioFile)
    {
        ArgumentNullException.ThrowIfNull(audioFile);
        Console.OutputEncoding = Encoding.UTF8;
        using var fileStream = audioFile.OpenReadStream();
        var speechConfig = SpeechConfig.FromSubscription(_speechKey, _speechRegion);
        byte[] audioData;
        using (var memoryStream = new MemoryStream())
        {
            await audioFile.CopyToAsync(memoryStream);
            audioData = memoryStream.ToArray();
        }
        byte channels = 1;
        byte bitsPerSample = 16;
        uint samplesPerSecond = 16000;
        var audioFormat = AudioStreamFormat.GetWaveFormatPCM(samplesPerSecond, bitsPerSample, channels);
        var audioConfig = AudioConfig.FromStreamInput(new BytesAudioStream(a
相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 00:17:17