使用Azure Speech SDK为discord-speech-recognition添加转写功能无文本输出问题
为discord-speech-recognition添加Azure语音转文本功能失败问题
我正在尝试为npm库discord-speech-recognition添加基于Azure Speech SDK的语音转文本功能,但无论是调用API还是使用SDK都没得到预期结果。目前仅能获取音频缓冲长度,无法得到转写文本。
实现代码
const sdk = require("microsoft-cognitiveservices-speech-sdk"); function getAzureRequestOptions(options) { const speechConfig = sdk.SpeechConfig.fromSubscription(options.key, options.region); if (options.lang) { speechConfig.speechRecognitionLanguage = options.lang; } return speechConfig; } function resolveSpeechWithAzureSpeechToText(audioBuffer, options) { console.log('resolveSpeechWithAzureSpeechToText called with audioBuffer length:', audioBuffer.length); const pushStream = sdk.AudioInputStream.createPushStream(); pushStream.write(audioBuffer); pushStream.close(); const speechConfig = getAzureRequestOptions(options); const audioConfig = sdk.AudioConfig.fromStreamInput(pushStream); const recognizer = new sdk.SpeechRecognizer(speechConfig, audioConfig); return new Promise((resolve, reject) => { console.log('Starting recognition...'); recognizer.recognizeOnceAsync( (result) => { console.log('Recognition succeeded'); recognizer.close(); resolve(result.text); console.log(result); }, (err) => { console.log('Recognition failed'); recognizer.close(); reject(err); console.log(err); } ); }); } module.exports.resolveSpeechWithAzureSpeechToText = resolveSpeechWithAzureSpeechToText;
控制台输出截图

问题分析与修复方案
1. 音频流关闭时机错误
你在写入音频缓冲后立刻关闭了pushStream,但recognizeOnceAsync还未完成音频读取操作,导致Azure SDK无法获取完整的音频数据。应该在识别完成后再关闭流。
2. 音频格式不匹配
Discord输出的音频通常是PCM 16位、48kHz单声道格式,而Azure SDK需要明确匹配的音频格式参数,否则无法正确解析音频数据。
3. 日志打印顺序问题
原代码中先执行resolve/reject再打印结果/错误,可能导致无法及时看到完整的调试信息,应该先打印再处理Promise状态。
修改后的代码
const sdk = require("microsoft-cognitiveservices-speech-sdk"); function getAzureRequestOptions(options) { const speechConfig = sdk.SpeechConfig.fromSubscription(options.key, options.region); if (options.lang) { speechConfig.speechRecognitionLanguage = options.lang; } return speechConfig; } function resolveSpeechWithAzureSpeechToText(audioBuffer, options) { console.log('resolveSpeechWithAzureSpeechToText called with audioBuffer length:', audioBuffer.length); // 明确设置音频格式,匹配Discord的PCM 48kHz单声道16位参数 const audioFormat = sdk.AudioStreamFormat.getWaveFormatPCM(48000, 16, 1); const pushStream = sdk.AudioInputStream.createPushStream(audioFormat); pushStream.write(audioBuffer); // 暂时不关闭流,等待识别完成后操作 const speechConfig = getAzureRequestOptions(options); const audioConfig = sdk.AudioConfig.fromStreamInput(pushStream); const recognizer = new sdk.SpeechRecognizer(speechConfig, audioConfig); return new Promise((resolve, reject) => { console.log('Starting recognition...'); recognizer.recognizeOnceAsync( (result) => { console.log('Recognition succeeded'); console.log(result); // 先打印结果再处理 recognizer.close(); pushStream.close(); // 识别完成后关闭流 resolve(result.text); }, (err) => { console.log('Recognition failed'); console.log(err); // 先打印错误再处理 recognizer.close(); pushStream.close(); reject(err); } ); }); } module.exports.resolveSpeechWithAzureSpeechToText = resolveSpeechWithAzureSpeechToText;
额外检查项
- 确认Azure Speech资源的密钥、区域配置正确,且资源处于可用状态;
- 验证Discord传入的
audioBuffer是否为原始PCM数据,如果是Opus编码格式,需要先解码为PCM再传给Azure SDK; - 可以将
audioBuffer保存为本地WAV文件,通过Azure Speech Studio测试该文件是否能正常转写,排除音频数据本身的问题。
内容的提问来源于stack exchange,提问作者Kamykalaur
相关产品推荐
相关产品推荐

