如何基于Angular、ASP.NET Core与Azure语音服务实现实时语音转文字
实时语音转文字(Azure Speech Service + WebSocket)实现方案
问题描述
我正在尝试用Microsoft Azure Speech Service构建实时语音转文字Web应用。目前通过MediaRecorder录制用户语音,录制完成后再发送到后端转成文本,但找不到完整示例说明如何从浏览器实时采集语音并通过WebSocket发送到后端。
现有代码
现有Angular代码(录制后上传)
import { Component, OnInit, OnDestroy, NgZone } from '@angular/core'; import { DomSanitizer, SafeUrl } from '@angular/platform-browser'; import * as WavEncoder from 'wav-encoder'; import { AdminService } from '../admin.service'; @Component({ selector: 'app-audio-recorder', templateUrl: './audio-recorder.component.html', styleUrls: ['./audio-recorder.component.css'] }) export class AudioRecorderComponent implements OnInit, OnDestroy { private mediaRecorder: MediaRecorder; private audioChunks: Blob[] = []; audioUrl: SafeUrl; isRecording = false; isLoading = false; transcript: { text: string, duration: string } = { text: '', duration: '' } constructor( private sanitizer: DomSanitizer, private adminService: AdminService, private ngZone: NgZone) { } ngOnInit(): void { navigator.mediaDevices.getUserMedia({ audio: true }) .then(stream => { this.mediaRecorder = new MediaRecorder(stream); this.mediaRecorder.ondataavailable = event => { if (event.data.size > 0) { this.audioChunks.push(event.data); } }; this.mediaRecorder.onstop = async () => { const audioBlob = new Blob(this.audioChunks, { type: 'audio/webm' }); const arrayBuffer = await audioBlob.arrayBuffer(); const audioBuffer = await this.decodeAudioData(arrayBuffer); const wavBlob = await this.convertToWav(audioBuffer); const audioURL = URL.createObjectURL(wavBlob); this.ngZone.run(() => { this.audioUrl = this.sanitizer.bypassSecurityTrustUrl(audioURL); }); this.ngZone.run(() => { this.isLoading = true; }); this.adminService.transcriptFile(wavBlob, "file") .subscribe({ next: (data) => { this.ngZone.run(() => { this.transcript.text = data this.isLoading = false }); }, error: (err) => { console.error(err) this.isLoading = false } }) this.audioChunks = []; }; }) .catch(error => { console.error('Error accessing microphone:', error); }); } startRecording() { if (this.mediaRecorder && this.mediaRecorder.state !== 'recording') { this.isRecording = true this.mediaRecorder.start(); } else { console.warn('MediaRecorder is not available or already recording'); } } stopRecording() { if (this.mediaRecorder && this.mediaRecorder.state === 'recording') { this.isRecording = false this.mediaRecorder.stop(); } else { console.warn('MediaRecorder is not recording'); } } ngOnDestroy() { if (this.audioUrl) { URL.revokeObjectURL(this.audioUrl as string); } } // Helper function to decode audio data private async decodeAudioData(arrayBuffer: ArrayBuffer): Promise<AudioBuffer> { const audioContext = new AudioContext(); return await audioContext.decodeAudioData(arrayBuffer); } // Helper function to convert AudioBuffer to WAV format private async convertToWav(audioBuffer: AudioBuffer): Promise<Blob> { const wavData = await WavEncoder.encode({ sampleRate: 16000, // 16 kHz bitDepth: 16, // 16 bits channelData: [ this.downsampleBuffer(audioBuffer.getChannelData(0), audioBuffer.sampleRate, 16000) ] }); return new Blob([wavData], { type: 'audio/wav' }); } // Helper function to downsample audio buffer private downsampleBuffer(buffer: Float32Array, sampleRate: number, targetRate: number): Float32Array { if (sampleRate === targetRate) { return buffer; } const sampleRateRatio = sampleRate / targetRate; const newLength = Math.round(buffer.length / sampleRateRatio); const result = new Float32Array(newLength); let offsetResult = 0; let offsetBuffer = 0; while (offsetResult < result.length) { const nextOffsetBuffer = Math.round((offsetResult + 1) * sampleRateRatio); let accum = 0, count = 0; for (let i = offsetBuffer; i < nextOffsetBuffer && i < buffer.length; i++) { accum += buffer[i]; count++; } result[offsetResult] = accum / count; offsetResult++; offsetBuffer = nextOffsetBuffer; } return result; } }
现有C#后端代码(批量转写)
[Route("api/[controller]")] [ApiController] public class AudioController : ControllerBase { [HttpPost("Transcript")] public async Task<string> Transcript(IFormFile audioFile) { ArgumentNullException.ThrowIfNull(audioFile); using var audioStream = audioFile.OpenReadStream(); Console.OutputEncoding = Encoding.UTF8; var audioKey = "..."; var audioRegion = "..."; using var fileStream = audioFile.OpenReadStream(); var speechConfig = SpeechConfig.FromSubscription(audioKey, audioRegion); byte[] audioData; using (var memoryStream = new MemoryStream()) { await audioFile.CopyToAsync(memoryStream); audioData = memoryStream.ToArray(); } byte channels = 1; byte bitsPerSample = 16; uint samplesPerSecond = 16000; // or 8000 based on your audio file's sample rate var audioFormat = AudioStreamFormat.GetWaveFormatPCM(samplesPerSecond, bitsPerSample, channels); var audioConfig = AudioConfig.FromStreamInput(new BytesAudioStream(audioData), audioFormat); speechConfig.SpeechRecognitionLanguage = "ar-SA"; var autoDetectSourceLanguageConfig = AutoDetectSourceLanguageConfig.FromLanguages(["ar-SA"]); var speechRecognizer = new SpeechRecognizer(speechConfig, autoDetectSourceLanguageConfig, audioConfig); Stopwatch stopwatch = Stopwatch.StartNew(); var speechRecognitionResult = await speechRecognizer.RecognizeOnceAsync(); stopwatch.Stop(); var duration = stopwatch.Elapsed; return speechRecognitionResult.Text; } } public class BytesAudioStream(byte[] audioData) : PullAudioInputStreamCallback { private readonly MemoryStream memoryStream = new(audioData); public override int Read(byte[] buffer, uint size) { return memoryStream.Read(buffer, 0, (int)size); } public override void Close() { memoryStream.Close(); } }
实时WebSocket传输实现方案
一、前端Angular修改:实时采集并发送音频数据
替换原有录制后上传的逻辑,使用AudioContext直接获取原始音频流,转换为Azure要求的16kHz、16位单声道PCM格式,通过WebSocket实时推送:
import { Component, OnInit, OnDestroy } from '@angular/core'; @Component({ selector: 'app-realtime-audio-recorder', templateUrl: './realtime-audio-recorder.component.html', styleUrls: ['./realtime-audio-recorder.component.css'] }) export class RealtimeAudioRecorderComponent implements OnInit, OnDestroy { private audioContext: AudioContext; private mediaStream: MediaStream; private scriptProcessor: ScriptProcessorNode; private ws: WebSocket; isRecording = false; transcript = ''; ngOnInit(): void { // 初始化WebSocket连接 this.ws = new WebSocket('ws://localhost:5000/api/audio/realtime-transcript'); this.ws.onmessage = (event) => { // 接收后端返回的实时识别结果 this.transcript = event.data; }; this.ws.onerror = (error) => { console.error('WebSocket错误:', error); }; this.ws.onclose = () => { console.log('WebSocket连接关闭'); this.stopRecording(); }; } async startRecording() { if (this.isRecording) return; this.isRecording = true; this.transcript = ''; // 获取麦克风流,强制指定16kHz采样率、单声道 this.mediaStream = await navigator.mediaDevices.getUserMedia({ audio: { sampleRate: 16000, channelCount: 1, echoCancellation: true } }); this.audioContext = new AudioContext({ sampleRate: 16000 }); const source = this.audioContext.createMediaStreamSource(this.mediaStream); // 创建脚本处理器,每次处理4096帧音频数据 this.scriptProcessor = this.audioContext.createScriptProcessor(4096, 1, 1); source.connect(this.scriptProcessor); this.scriptProcessor.connect(this.audioContext.destination); // 实时处理并发送音频数据 this.scriptProcessor.onaudioprocess = (event) => { if (!this.isRecording || this.ws.readyState !== WebSocket.OPEN) return; // 获取左声道的Float32格式音频数据 const inputBuffer = event.inputBuffer.getChannelData(0); // 转换为Azure要求的16位PCM格式(Uint8Array) const pcmData = this.float32ToInt16(inputBuffer); // 通过WebSocket发送二进制数据 this.ws.send(pcmData); }; } stopRecording() { this.isRecording = false; // 关闭音频流和上下文 this.mediaStream?.getTracks().forEach(track => track.stop()); this.audioContext?.close(); this.scriptProcessor?.disconnect(); // 关闭WebSocket连接 if (this.ws.readyState === WebSocket.OPEN) { this.ws.close(); } } ngOnDestroy() { this.stopRecording(); } // 将Float32音频数据转换为16位PCM格式(小端序) private float32ToInt16(buffer: Float32Array): Uint8Array { const length = buffer.length; const result = new Uint8Array(length * 2); let index = 0; for (let i = 0; i < length; i++) { let sample = buffer[i]; // 限制音频范围在[-1, 1] sample = Math.max(-1, Math.min(1, sample)); // 转换为16位整数(范围:-32768 到 32767) sample = sample < 0 ? sample * 32768 : sample * 32767; // 写入Uint8Array(小端序存储) result[index++] = sample & 0xff; result[index++] = (sample >> 8) & 0xff; } return result; } }
二、后端C#修改:WebSocket服务器 + Azure实时语音识别
在ASP.NET Core中启用WebSocket中间件,接收前端推送的音频数据,实时转发至Azure Speech Service进行识别,并将结果回传前端:
1. 启用WebSocket中间件(Program.cs)
var builder = WebApplication.CreateBuilder(args); builder.Services.AddControllers(); builder.Services.AddEndpointsApiExplorer(); builder.Services.AddSwaggerGen(); var app = builder.Build(); // 启用WebSocket中间件 app.UseWebSockets(); if (app.Environment.IsDevelopment()) { app.UseSwagger(); app.UseSwaggerUI(); } app.UseHttpsRedirection(); app.UseAuthorization(); app.MapControllers(); app.Run();
2. 实现WebSocket实时识别接口(AudioController)
using Microsoft.AspNetCore.Mvc; using Microsoft.CognitiveServices.Speech; using Microsoft.CognitiveServices.Speech.Audio; using System.Buffers; using System.Net.WebSockets; using System.Text; [Route("api/[controller]")] [ApiController] public class AudioController : ControllerBase { private readonly string _speechKey = "你的Azure Speech密钥"; private readonly string _speechRegion = "你的Azure区域(如eastus)"; [HttpGet("realtime-transcript")] public async Task RealtimeTranscript() { if (!HttpContext.WebSockets.IsWebSocketRequest) { HttpContext.Response.StatusCode = StatusCodes.Status400BadRequest; return; } using var webSocket = await HttpContext.WebSockets.AcceptWebSocketAsync(); // 复用内存缓冲区,减少GC开销 var buffer = ArrayPool<byte>.Shared.Rent(4096 * 2); try { // 配置Azure Speech服务 var speechConfig = SpeechConfig.FromSubscription(_speechKey, _speechRegion); speechConfig.SpeechRecognitionLanguage = "ar-SA"; // 创建Push类型音频输入流,用于实时接收音频数据 using var audioInputStream = AudioInputStream.CreatePushStream(); using var audioConfig = AudioConfig.FromStreamInput(audioInputStream); // 创建实时语音识别器 using var speechRecognizer = new SpeechRecognizer(speechConfig, audioConfig); // 订阅实时识别中间结果事件 speechRecognizer.Recognizing += async (s, e) => { if (e.Result.Reason == ResultReason.RecognizingSpeech) { var message = Encoding.UTF8.GetBytes(e.Result.Text); await webSocket.SendAsync( new ArraySegment<byte>(message, 0, message.Length), WebSocketMessageType.Text, true, CancellationToken.None ); } }; // 订阅最终识别结果事件 speechRecognizer.Recognized += async (s, e) => { if (e.Result.Reason == ResultReason.RecognizedSpeech) { var message = Encoding.UTF8.GetBytes(e.Result.Text); await webSocket.SendAsync( new ArraySegment<byte>(message, 0, message.Length), WebSocketMessageType.Text, true, CancellationToken.None ); } else if (e.Result.Reason == ResultReason.NoMatch) { var message = Encoding.UTF8.GetBytes("未识别到有效语音"); await webSocket.SendAsync( new ArraySegment<byte>(message, 0, message.Length), WebSocketMessageType.Text, true, CancellationToken.None ); } }; // 启动连续识别 await speechRecognizer.StartContinuousRecognitionAsync(); // 循环接收WebSocket的音频数据 WebSocketReceiveResult result; do { result = await webSocket.ReceiveAsync(new ArraySegment<byte>(buffer), CancellationToken.None); if (result.MessageType == WebSocketMessageType.Binary) { // 将接收到的PCM数据推送到Azure Speech流 audioInputStream.Write(buffer, 0, result.Count); } } while (!result.CloseStatus.HasValue); // 停止识别 await speechRecognizer.StopContinuousRecognitionAsync(); } catch (Exception ex) { Console.Error.WriteLine($"实时识别错误:{ex.Message}"); } finally { // 归还内存缓冲区 ArrayPool<byte>.Shared.Return(buffer); await webSocket.CloseAsync(WebSocketCloseStatus.NormalClosure, "连接正常关闭", CancellationToken.None); } } // 保留原有批量转写接口(可选) [HttpPost("Transcript")] public async Task<string> Transcript(IFormFile audioFile) { ArgumentNullException.ThrowIfNull(audioFile); Console.OutputEncoding = Encoding.UTF8; using var fileStream = audioFile.OpenReadStream(); var speechConfig = SpeechConfig.FromSubscription(_speechKey, _speechRegion); byte[] audioData; using (var memoryStream = new MemoryStream()) { await audioFile.CopyToAsync(memoryStream); audioData = memoryStream.ToArray(); } byte channels = 1; byte bitsPerSample = 16; uint samplesPerSecond = 16000; var audioFormat = AudioStreamFormat.GetWaveFormatPCM(samplesPerSecond, bitsPerSample, channels); var audioConfig = AudioConfig.FromStreamInput(new BytesAudioStream(a
相关产品推荐
相关产品推荐

