You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Azure Speech SDK为discord-speech-recognition添加转写功能无文本输出问题

为discord-speech-recognition添加Azure语音转文本功能失败问题

我正在尝试为npm库discord-speech-recognition添加基于Azure Speech SDK的语音转文本功能,但无论是调用API还是使用SDK都没得到预期结果。目前仅能获取音频缓冲长度,无法得到转写文本。

实现代码

const sdk = require("microsoft-cognitiveservices-speech-sdk");

function getAzureRequestOptions(options) {
    const speechConfig = sdk.SpeechConfig.fromSubscription(options.key, options.region);
    if (options.lang) {
        speechConfig.speechRecognitionLanguage = options.lang;
    }
    return speechConfig;
}

function resolveSpeechWithAzureSpeechToText(audioBuffer, options) {
    console.log('resolveSpeechWithAzureSpeechToText called with audioBuffer length:', audioBuffer.length);
    const pushStream = sdk.AudioInputStream.createPushStream();

    pushStream.write(audioBuffer);
    pushStream.close();

    const speechConfig = getAzureRequestOptions(options);
    const audioConfig = sdk.AudioConfig.fromStreamInput(pushStream);
    const recognizer = new sdk.SpeechRecognizer(speechConfig, audioConfig);

    return new Promise((resolve, reject) => {
        console.log('Starting recognition...');
        recognizer.recognizeOnceAsync(
            (result) => {
                console.log('Recognition succeeded');
                recognizer.close();
                resolve(result.text);
                console.log(result);
            },
            (err) => {
                console.log('Recognition failed');
                recognizer.close();
                reject(err);
                console.log(err);
            }
        );
    });
}

module.exports.resolveSpeechWithAzureSpeechToText = resolveSpeechWithAzureSpeechToText;

控制台输出截图

控制台输出截图


问题分析与修复方案

1. 音频流关闭时机错误

你在写入音频缓冲后立刻关闭了pushStream,但recognizeOnceAsync还未完成音频读取操作,导致Azure SDK无法获取完整的音频数据。应该在识别完成后再关闭流。

2. 音频格式不匹配

Discord输出的音频通常是PCM 16位、48kHz单声道格式,而Azure SDK需要明确匹配的音频格式参数,否则无法正确解析音频数据。

3. 日志打印顺序问题

原代码中先执行resolve/reject再打印结果/错误,可能导致无法及时看到完整的调试信息,应该先打印再处理Promise状态。

修改后的代码

const sdk = require("microsoft-cognitiveservices-speech-sdk");

function getAzureRequestOptions(options) {
    const speechConfig = sdk.SpeechConfig.fromSubscription(options.key, options.region);
    if (options.lang) {
        speechConfig.speechRecognitionLanguage = options.lang;
    }
    return speechConfig;
}

function resolveSpeechWithAzureSpeechToText(audioBuffer, options) {
    console.log('resolveSpeechWithAzureSpeechToText called with audioBuffer length:', audioBuffer.length);
    
    // 明确设置音频格式,匹配Discord的PCM 48kHz单声道16位参数
    const audioFormat = sdk.AudioStreamFormat.getWaveFormatPCM(48000, 16, 1);
    const pushStream = sdk.AudioInputStream.createPushStream(audioFormat);

    pushStream.write(audioBuffer);
    // 暂时不关闭流,等待识别完成后操作

    const speechConfig = getAzureRequestOptions(options);
    const audioConfig = sdk.AudioConfig.fromStreamInput(pushStream);
    const recognizer = new sdk.SpeechRecognizer(speechConfig, audioConfig);

    return new Promise((resolve, reject) => {
        console.log('Starting recognition...');
        recognizer.recognizeOnceAsync(
            (result) => {
                console.log('Recognition succeeded');
                console.log(result); // 先打印结果再处理
                recognizer.close();
                pushStream.close(); // 识别完成后关闭流
                resolve(result.text);
            },
            (err) => {
                console.log('Recognition failed');
                console.log(err); // 先打印错误再处理
                recognizer.close();
                pushStream.close();
                reject(err);
            }
        );
    });
}

module.exports.resolveSpeechWithAzureSpeechToText = resolveSpeechWithAzureSpeechToText;

额外检查项

  • 确认Azure Speech资源的密钥、区域配置正确,且资源处于可用状态;
  • 验证Discord传入的audioBuffer是否为原始PCM数据,如果是Opus编码格式,需要先解码为PCM再传给Azure SDK;
  • 可以将audioBuffer保存为本地WAV文件,通过Azure Speech Studio测试该文件是否能正常转写,排除音频数据本身的问题。

内容的提问来源于stack exchange,提问作者Kamykalaur

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.04 13:55:14