You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Azure语音服务JS SDK实现发音评估时遭遇"Could not deserialize speech context"错误的求助

使用Azure语音服务JS SDK实现发音评估时遭遇"Could not deserialize speech context"错误的求助

我正在尝试基于Azure语音服务的JS SDK开发发音评估功能,结果控制台持续弹出这个错误:

"Could not deserialize speech context. websocket error code: 1007"

我已经确认待处理的WAV录音文件是正常可用的,下面是我的具体实现代码:

发音评估核心逻辑

assessPronunciation(fileUrl) {
    const speechConfig = window.SpeechSDK.SpeechConfig.fromSubscription("xxx", "westeurope");
    speechConfig.speechRecognitionLanguage = "en-GB";

    // Fetch the WAV file and create an AudioConfig
    fetch(fileUrl)
      .then(response => response.blob())
      .then(blob => {
        // Convert the blob to a File object
        const file = new File([blob], "audio.wav", { type: "audio/wav" });

        // Create an AudioConfig using the File object
        const audioConfig = window.SpeechSDK.AudioConfig.fromWavFileInput(file);

        var pronunciationAssessmentConfig = new window.SpeechSDK.PronunciationAssessmentConfig({
          referenceText: "Hello this is a test",
          gradingSystem: "HundredMark",
          granularity: "Phoneme"
        });

        var speechRecognizer = new window.SpeechSDK.SpeechRecognizer(speechConfig, audioConfig);

        pronunciationAssessmentConfig.applyTo(speechRecognizer);

        speechRecognizer.sessionStarted = (s, e) => {
          console.log(`SESSION ID: ${e.sessionId}`);
        };
        pronunciationAssessmentConfig.applyTo(speechRecognizer);
        
        speechRecognizer.recognizeOnceAsync(
          function(speechRecognitionResult) {
            if (speechRecognitionResult.reason === window.SpeechSDK.ResultReason.RecognizedSpeech) {
              // The pronunciation assessment result as a Speech SDK object
              var pronunciationAssessmentResult = SpeechSDK.PronunciationAssessmentResult.fromResult(speechRecognitionResult);
              console.log("pronunciationAssessmentResult", pronunciationAssessmentResult);
          
              // The pronunciation assessment result as a JSON string
              var pronunciationAssessmentResultJson = speechRecognitionResult.properties.getProperty(SpeechSDK.PropertyId.SpeechServiceResponse_JsonResult);
              console.log("pronunciationAssessmentResultJson", pronunciationAssessmentResultJson);
            } else {
              console.error("Speech not recognized. Reason:", speechRecognitionResult);
            }
          },
          function(error) {
            console.error("Error during recognition:", error);
            if (error instanceof window.SpeechSDK.SpeechRecognitionCanceledEventArgs) {
              console.error("Recognition canceled. Reason:", error.reason);
              console.error("Error details:", error.errorDetails);
            }
          }
        );
      })
      .catch(error => {
        console.error("Error fetching WAV file:", error);
      });
  }

录音配置代码

startRecording(event) {
    event.preventDefault();
    if (navigator.mediaDevices && navigator.mediaDevices.getUserMedia) {
      navigator.mediaDevices.getUserMedia({ audio: true }).then(stream => {
        this.recorder = new RecordRTC(stream, {
          type: 'audio',
          mimeType: 'audio/wav',
          recorderType: RecordRTC.StereoAudioRecorder,
          desiredSampRate: 16000,
          numberOfAudioChannels: 1,
          audioBitsPerSecond: 128000
        });
        this.startRecorder(event);
      }).catch((error) => {
        console.log("The following error occurred: " + error);
        alert("Please grant permission for microphone access");
      });
    } else {
      alert("Your browser does not support audio recording, please use a different browser or update your current browser");
    }
  }

解决方案

经过调试,我发现问题出在发音评估配置的初始化方式和音频输入的处理逻辑上。修正后的核心代码如下:

// 替换原有的fromWavFileInput,改用流输入创建AudioConfig
var audioConfig = window.SpeechSDK.AudioConfig.fromStreamInput(pushStream);

// 调整PronunciationAssessmentConfig的创建方式,使用SDK内置枚举而非字符串参数
var pronunciationAssessmentConfig = new window.SpeechSDK.PronunciationAssessmentConfig(
      "My voice is my passport, verify me.",
      window.SpeechSDK.PronunciationAssessmentGradingSystem.HundredMark,
      window.SpeechSDK.PronunciationAssessmentGranularity.Phoneme
  );

关键修改点说明

  1. 音频输入方式:将fromWavFileInput改为fromStreamInput,避免文件输入在序列化时出现的上下文不兼容问题
  2. 配置初始化:使用SDK提供的枚举类型(如PronunciationAssessmentGradingSystem.HundredMark)替代直接传入字符串,确保参数格式完全符合服务端的序列化要求,从根源上避免"无法反序列化语音上下文"的错误

备注:内容来源于stack exchange,提问作者nico_lrx

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.14 15:13:09