Azure Speech-to-text Python:start_continuous_recognition无法在流结束时停止
Azure Speech-to-text 连续识别延迟问题
我用Python调用Azure Speech-to-text服务,输入data是仅几秒的音频字节串,预期云服务在流结束时停止处理并返回识别文本,但实际recognized事件触发延迟了约5分钟。
代码实现
speech_config = speechsdk.SpeechConfig(subscription=API_KEY, region="westeurope", speech_recognition_language='de-DE') stream = PushAudioInputStream(stream_format= AudioStreamFormat(samples_per_second=sample_rate, bits_per_sample=SAMPLE_WIDTH * 8, compressed_stream_format=speechsdk.AudioStreamContainerFormat.FLAC)) audio_input = speechsdk.AudioConfig(stream=stream) stream.write(data) speech_recognizer = speechsdk.SpeechRecognizer(speech_config=speech_config, audio_config=audio_input) speech_recognizer.start_continuous_recognition() done = False def stop_recognition(evt): logger.debug("Stopped MS Azure recognition: %s", evt) nonlocal done done = True def recognized(evt): logger.info("Recognized MS Azure transcript: %s", evt) nonlocal text text += " " + evt.result.text speech_recognizer.recognizing.connect(lambda evt: print('RECOGNIZING: {}'.format(evt))) speech_recognizer.recognized.connect(lambda evt: print('RECOGNIZED: {}'.format(evt))) speech_recognizer.session_started.connect(lambda evt: print('SESSION STARTED: {}'.format(evt))) speech_recognizer.session_stopped.connect(lambda evt: print('SESSION STOPPED {}'.format(evt))) speech_recognizer.canceled.connect(lambda evt: print('CANCELED {}'.format(evt))) speech_recognizer.recognized.connect(recognized) speech_recognizer.session_stopped.connect(stop_recognition) speech_recognizer.canceled.connect(stop_recognition) while not done: time.sleep(.5) speech_recognizer.stop_continuous_recognition()
延迟日志
2022-11-13 23:58:19,504 - speech_processing.speech_recognition.speech_recognition - DEBUG - Sending 192000 bytes (6 sec) for recognition RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=50e5c478cdc34e0a8ced3867be493bc3, text="telefon", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=d1448833ac8f40ef9c1ebc4cae488bcd, text="telefonspeicher", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=cdf9f074c13b4a2c94960ec147db765c, text="telefon speichere", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=548133156bb44dc8ae08fd0848fa8ec5, text="telefon speichere als", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=c03970619f1e42278b2a2ef19ee4f1fe, text="telefon speichere als bärbel", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=ff5f6a18d1e4409cab2661582cb8a693, text="telefon speichere als bärbel 0", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=a3cb8f82c62b4235abc2fea2696342f8, text="telefon speichere als bärbel 03", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=cafc805031654aa4865a4fe1b742d1cd, text="telefon speichere als bärbel 038", reason=ResultReason.RecognizingSpeech)) RECOGNIZING: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=69782c9485244e3191846b924adb3807, text="telefon speichere als bärbel 0385", reason=ResultReason.RecognizingSpeech)) RECOGNIZED: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=9d92890d52d84b7f926a6977d6324ca1, text="Telefon speichere als Bärbel 0385.", reason=ResultReason.RecognizedSpeech)) 2022-11-14 00:03:26,487 - speech_processing.speech_recognition.speech_recognition - INFO - Recognized MS Azure transcript: SpeechRecognitionEventArgs(session_id=2e4c92f4fed6498f8f5260199bdcc5d7, result=SpeechRecognitionResult(result_id=9d92890d52d84b7f926a6977d6324ca1, text="Telefon speichere als Bärbel 0385.", reason=ResultReason.RecognizedSpeech))
问题原因&解决方案
核心原因
你使用连续识别模式但未明确告知Azure服务音频流已结束。Azure Speech服务在连续识别模式下会持续等待更多音频输入,直到收到流结束信号或触发超时(默认约5分钟),这就是延迟的根源。同时你写入音频数据后未关闭流,服务无法判断音频是否全部发送完成。
修复步骤
- 关闭音频流:在
stream.write(data)后调用stream.close(),明确告知服务音频已全部发送。 - 切换识别模式(可选):如果处理的都是短音频片段,使用单次识别模式(
recognize_once())更高效,服务会在处理完音频后立即返回结果,无需等待超时。
修复后的连续识别模式代码
speech_config = speechsdk.SpeechConfig(subscription=API_KEY, region="westeurope", speech_recognition_language='de-DE') stream = PushAudioInputStream(stream_format= AudioStreamFormat(samples_per_second=sample_rate, bits_per_sample=SAMPLE_WIDTH * 8, compressed_stream_format=speechsdk.AudioStreamContainerFormat.FLAC)) audio_input = speechsdk.AudioConfig(stream=stream) stream.write(data) # 关键:关闭流,告知服务音频传输完成 stream.close() speech_recognizer = speechsdk.SpeechRecognizer(speech_config=speech_config, audio_config=audio_input) speech_recognizer.start_continuous_recognition() done = False def stop_recognition(evt): logger.debug("Stopped MS Azure recognition: %s", evt) nonlocal done done = True def recognized(evt): logger.info("Recognized MS Azure transcript: %s", evt) nonlocal text text += " " + evt.result.text speech_recognizer.recognizing.connect(lambda evt: print('RECOGNIZING: {}'.format(evt))) speech_recognizer.recognized.connect(lambda evt: print('RECOGNIZED: {}'.format(evt))) speech_recognizer.session_started.connect(lambda evt: print('SESSION STARTED: {}'.format(evt))) speech_recognizer.session_stopped.connect(lambda evt: print('SESSION STOPPED {}'.format(evt))) speech_recognizer.canceled.connect(lambda evt: print('CANCELED {}'.format(evt))) speech_recognizer.recognized.connect(recognized) speech_recognizer.session_stopped.connect(stop_recognition) speech_recognizer.canceled.connect(stop_recognition) while not done: time.sleep(.5) speech_recognizer.stop_continuous_recognition()
单次识别模式替代方案(适合短音频)
speech_config = speechsdk.SpeechConfig(subscription=API_KEY, region="westeurope", speech_recognition_language='de-DE') stream = PushAudioInputStream(stream_format= AudioStreamFormat(samples_per_second=sample_rate, bits_per_sample=SAMPLE_WIDTH * 8, compressed_stream_format=speechsdk.AudioStreamContainerFormat.FLAC)) audio_input = speechsdk.AudioConfig(stream=stream) stream.write(data) stream.close() speech_recognizer = speechsdk.SpeechRecognizer(speech_config=speech_config, audio_config=audio_input) # 单次识别直接获取结果 result = speech_recognizer.recognize_once() if result.reason == speechsdk.ResultReason.RecognizedSpeech: print(f"识别结果: {result.text}") logger.info(f"Recognized MS Azure transcript: {result.text}") elif result.reason == speechsdk.ResultReason.NoMatch: print(f"未识别到语音: {result.no_match_details}") elif result.reason == speechsdk.ResultReason.Canceled: cancellation_details = result.cancellation_details print(f"识别被取消: {cancellation_details.reason}") if cancellation_details.reason == speechsdk.CancellationReason.Error: print(f"错误详情: {cancellation_details.error_details}")
验证效果
修复后,服务会在收到流关闭信号后立即处理完剩余音频,触发recognized事件并结束会话,不会再出现5分钟的延迟。
内容的提问来源于stack exchange,提问作者k_o_
相关产品推荐
相关产品推荐

