You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

.NET环境下Azure语音识别重复输出问题排查求助

问题:实时语音转文字单句重复输出

我用.NET基于Azure搭建麦克风实时语音转文字功能时,遇到了单句重复输出的问题。比如对着麦克风说"Hello my name is John",代码会重复输出这句话数十次。我需要实现像Azure Speech Studio那样,单句仅输出一次的效果。

以下是我的代码:

Program.cs

using Microsoft.AspNetCore.Hosting;
using Microsoft.Extensions.Configuration;
using Microsoft.Extensions.DependencyInjection;
using Microsoft.Extensions.Hosting;
using Microsoft.Extensions.Logging;
using System;
using System.Linq;
using [ ].Hubs;
using [ ].Services;
using [ ].Data;

public class Program{
    public static void Main(string[] args)
    {
        CreateHostBuilder(args).Build().Run();
    }

    public static IHostBuilder CreateHostBuilder(string[] args) =>
        Host.CreateDefaultBuilder(args)
            .ConfigureWebHostDefaults(webBuilder =>
            {
                webBuilder.UseUrls("http://localhost:5001");
                webBuilder.ConfigureServices((context, services) =>
                {
                    services.AddControllers();
                    services.AddSignalR();

                    // 配置CORS
                    services.AddCors(options =>
                    {
                        options.AddDefaultPolicy(builder =>
                        {
                            builder.WithOrigins("http://localhost:3000")
                                   .AllowAnyHeader()
                                   .AllowAnyMethod()
                                   .AllowCredentials();
                        });
                    });

                    services.AddSingleton<SpeechService>();
                    services.AddSingleton<TranscriptionService>();

                    // 添加DbContext和托管服务
                    services.AddDbContext<CosmosDbContext>();
                    services.AddHostedService<CosmosDbTestService>();
                })
                .Configure((context, app) =>
                {
                    if (context.HostingEnvironment.IsDevelopment())
                    {
                        app.UseDeveloperExceptionPage();
                    }

                    // 启用CORS
                    app.UseCors();

                    app.UseRouting();

                    app.UseEndpoints(endpoints =>
                    {
                        endpoints.MapControllers();
                        endpoints.MapHub<TranscriptionHub>("/transcriptionHub");
                    });
                });
            })
            .ConfigureLogging(logging =>
            {
                logging.ClearProviders();
                logging.AddConsole();
            });
}

SpeechService.cs (/Service)

using Microsoft.CognitiveServices.Speech;
using Microsoft.CognitiveServices.Speech.Audio;

namespace [ ].Services
{
    public class SpeechService
    {
        private readonly SpeechRecognizer _speechRecognizer;
        private bool isRecognizing = false;
        public event Action<string>? OnRecognizing;
        public event Action<string>? OnRecognized;

        public SpeechService()
        {
            var subscriptionKey = Environment.GetEnvironmentVariable("AZURE_SPEECH_KEY");
            var region = Environment.GetEnvironmentVariable("AZURE_SPEECH_REGION");

            if (string.IsNullOrEmpty(subscriptionKey) || string.IsNullOrEmpty(region))
            {
                throw new InvalidOperationException("必须通过环境变量提供Azure语音服务的密钥和区域。");
            }

            var speechConfig = SpeechConfig.FromSubscription(subscriptionKey, region);
            speechConfig.SpeechRecognitionLanguage = "de-DE";
            speechConfig.EnableDictation(); // 启用听写模式以支持显式标点

            var audioConfig = AudioConfig.FromDefaultMicrophoneInput();
            _speechRecognizer = new SpeechRecognizer(speechConfig, audioConfig);

            _speechRecognizer.Recognizing += (s, e) =>
            {
                if (e.Result.Reason == ResultReason.RecognizingSpeech)
                {
                    OnRecognizing?.Invoke(e.Result.Text);
                }
            };

            _speechRecognizer.Recognized += (s, e) =>
            {
                if (e.Result.Reason == ResultReason.RecognizedSpeech)
                {
                    OnRecognized?.Invoke(e.Result.Text);
                }
            };

            _speechRecognizer.Canceled += (s, e) =>
            {
                isRecognizing = false;
                Console.WriteLine($"识别已取消: {e.Reason}, {e.ErrorDetails}");
            };

            _speechRecognizer.SessionStopped += (s, e) =>
            {
                isRecognizing = false;
                Console.WriteLine($"会话已停止: {e.SessionId}");
            };
        }

        public async Task StartRecognitionAsync()
        {
            if (!isRecognizing)
            {
                isRecognizing = true;
                await _speechRecognizer.StartContinuousRecognitionAsync().ConfigureAwait(false);
            }
        }

        public async Task StopRecognitionAsync()
        {
            if (isRecognizing)
            {
                await _speechRecognizer.StopContinuousRecognitionAsync().ConfigureAwait(false);
                isRecognizing = false;
            }
        }
    }
}

TranscriptionService.cs (/Service)

using Microsoft.AspNetCore.SignalR;
using System.Collections.Concurrent;
using [ ].Hubs;

namespace [ ].Services
{
    public class TranscriptionService
    {
        private readonly IHubContext<TranscriptionHub> _hubContext;
        private readonly ConcurrentDictionary<string, string> _connections = new ConcurrentDictionary<string, string>();

        public TranscriptionService(IHubContext<TranscriptionHub> hubContext)
        {
            _hubContext = hubContext;
        }

        public void AddConnection(string connectionId)
        {
            _connections[connectionId] = connectionId;
        }

        public void RemoveConnection(string connectionId)
        {
            _connections.TryRemove(connectionId, out _);
        }

        public async Task BroadcastRecognizing(string text)
        {
            foreach (var connectionId in _connections.Keys)
            {
                await _hubContext.Clients.Client(connectionId).SendAsync("ReceiveRecognizing", text);
            }
        }

        public async Task BroadcastRecognized(string text)
        {
            foreach (var connectionId in _connections.Keys)
            {
                await _hubContext.Clients.Client(connectionId).SendAsync("ReceiveRecognized", text);
            }
        }
    }
}

TranscriptionHub.cs (/Hub)

using Microsoft.AspNetCore.SignalR;
using [ ].Services;

namespace [ ].Hubs
{
    public class TranscriptionHub : Hub
    {
        private readonly SpeechService _speechService;
        private readonly TranscriptionService _transcriptionService;

        public TranscriptionHub(SpeechService speechService, TranscriptionService transcriptionService)
        {
            _speechService = speechService;
            _transcriptionService = transcriptionService;
        }

        public override async Task OnConnectedAsync()
        {
            _transcriptionService.AddConnection(Context.ConnectionId);
            _speechService.OnRecognizing += HandleRecognizing;
            _speechService.OnRecognized += HandleRecognized;
            await base.OnConnectedAsync();
        }

        public override async Task OnDisconnectedAsync(Exception? exception)
        {
            _transcriptionService.RemoveConnection(Context.ConnectionId);
            _speechService.OnRecognizing -= HandleRecognizing;
            _speechService.OnRecognized -= HandleRecognized;
            await base.OnDisconnectedAsync(exception);
        }

        private async void HandleRecognizing(string text)
        {
            await _transcriptionService.BroadcastRecognizing(text);
        }

        private async void HandleRecognized(string text)
        {
            await _transcriptionService.BroadcastRecognized(text);
        }

        public async Task StartTranscription()
        {
            await _speechService.StartRecognitionAsync();
        }

        public async Task StopTranscription()
        {
            await _speechService.StopRecognitionAsync();
        }
    }
}

解决方案

1. 修复事件重复订阅问题

你的SpeechService是单例服务,每次客户端连接TranscriptionHub时,都会给OnRecognizing和OnRecognized事件添加新的处理方法。当有多个客户端连接时,同一个识别结果会被多次触发并广播,导致重复输出。

修改方式:让SpeechService直接依赖TranscriptionService,内部完成广播逻辑,避免外部重复订阅事件。

修改后的SpeechService.cs

using Microsoft.CognitiveServices.Speech;
using Microsoft.CognitiveServices.Speech.Audio;

namespace [ ].Services
{
    public class SpeechService
    {
        private readonly SpeechRecognizer _speechRecognizer;
        private bool isRecognizing = false;
        private readonly TranscriptionService _transcriptionService;

        public SpeechService(TranscriptionService transcriptionService)
        {
            _transcriptionService = transcriptionService;
            var subscriptionKey = Environment.GetEnvironmentVariable("AZURE_SPEECH_KEY");
            var region = Environment.GetEnvironmentVariable("AZURE_SPEECH_REGION");

            if (string.IsNullOrEmpty(subscriptionKey) || string.IsNullOrEmpty(region))
            {
                throw new InvalidOperationException("必须通过环境变量提供Azure语音服务的密钥和区域。");
            }

            var speechConfig = SpeechConfig.FromSubscription(subscriptionKey, region);
            speechConfig.SpeechRecognitionLanguage = "de-DE";
            speechConfig.EnableDictation(); // 启用听写模式以支持显式标点

            var audioConfig = AudioConfig.FromDefaultMicrophoneInput();
            _speechRecognizer = new SpeechRecognizer(speechConfig, audioConfig);

            _speechRecognizer.Recognizing += async (s, e) =>
            {
                if (e.Result.Reason == ResultReason.RecognizingSpeech)
                {
                    await _transcriptionService.BroadcastRecognizing(e.Result.Text);
                }
            };

            _speechRecognizer.Recognized += async (s, e) =>
            {
                if (e.Result.Reason == ResultReason.RecognizedSpeech)
                {
                    await _transcriptionService.BroadcastRecognized(e.Result.Text);
                }
            };

            _speechRecognizer.Canceled += (s, e) =>
            {
                isRecognizing = false;
                Console.WriteLine($"识别已取消: {e.Reason}, {e.ErrorDetails}");
            };

            _speechRecognizer.SessionStopped += (s, e) =>
            {
                isRecognizing = false;
                Console.WriteLine($"会话已停止: {e.SessionId}");
            };
        }

        public async Task StartRecognitionAsync()
        {
            if (!isRecognizing)
            {
                isRecognizing = true;
                await _speechRecognizer.StartContinuousRecognitionAsync().ConfigureAwait(false);
            }
        }

        public async Task StopRecognitionAsync()
        {
            if (isRecognizing)
            {
                await _speechRecognizer.StopContinuousRecognitionAsync().ConfigureAwait(false);
                isRecognizing = false;
            }
        }
    }
}

修改后的TranscriptionHub.cs

移除事件订阅逻辑,因为SpeechService已经内部处理广播:

using Microsoft.AspNetCore.SignalR;
using [ ].Services;

namespace [ ].Hubs
{
    public class TranscriptionHub : Hub
    {
        private readonly SpeechService _speechService;
        private readonly TranscriptionService _transcriptionService;

        public TranscriptionHub(SpeechService speechService, TranscriptionService transcriptionService)
        {
            _speechService = speechService;
            _transcriptionService = transcriptionService;
        }

        public override async Task OnConnectedAsync()
        {
            _transcriptionService.AddConnection(Context.ConnectionId);
            await base.OnConnectedAsync();
        }

        public override async Task OnDisconnectedAsync(Exception? exception)
        {
            _transcriptionService.RemoveConnection(Context.ConnectionId);
            await base.OnDisconnectedAsync(exception);
        }

        public async Task StartTranscription()
        {
            await _speechService.StartRecognitionAsync();
        }

        public async Task StopTranscription()
        {
            await _speechService.StopRecognitionAsync();
        }
    }
}

2. 区分中间结果与最终结果

Azure语音服务的Recognizing事件返回的是实时中间识别结果(说话过程中会多次触发,文本逐步完善),而Recognized事件返回的是最终确认的单句结果(仅触发一次)。如果前端同时监听两个事件并追加内容,会出现重复的中间文本。

建议:

  • 若只需最终单句输出,前端仅监听ReceiveRecognized事件
  • 若需实时显示识别过程,前端监听ReceiveRecognizing时用新文本覆盖旧内容,而非追加

内容的提问来源于stack exchange,提问作者ecobiz

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 21:55:13