You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在React Native(Expo)中使用Azure Custom Commands API

问题描述

我正尝试在React Native中使用Azure Speech Services的Custom Commands,但找不到可用的JavaScript示例。已找到的资源均未具体展示Custom Commands的使用方式:

  • 某含JS代码的示例项目:未说明如何通过麦克风将音频发送至API
  • 微软官方文档:仅提供C#示例,自行编写的对应JS代码无法运行

目前已通过expo-av实现音频录制,但不知道如何将录制的音频转换为音频流并传递给Azure Speech SDK,当前代码如下:

import 'react-native-get-random-values';
import { useState, useEffect } from 'react';
import { Audio } from 'expo-av';
import { StyleSheet, Button, View } from 'react-native';
import * as SpeechSDK from 'microsoft-cognitiveservices-speech-sdk';

const APPLICATION_ID = '6e3b5880-918e-11ed-a04e-cf28b0773cf7';
const SPEECH_KEY = '51ecddfd91564074b3cb2cee2f9e4f52';
const SPEECH_RESOURCE_REGION = 'centralindia'

export default function App() {

  const [recording, setRecording] = useState();
  const [dialogueServiceConnector, setDialogueServiceConnector] = useState();
  const [lastRecordedSound, setLastRecordedSound] = useState();

  function initializeDialogueServiceConnector() {
    const commandsConfig = new SpeechSDK.CustomCommandsConfig.fromSubscription(APPLICATION_ID, SPEECH_KEY, SPEECH_RESOURCE_REGION);
    const connector = new SpeechSDK.DialogServiceConnector(commandsConfig);
    connector.activityReceived = (sender, arg) => {
      console.log(`Activity received, activity=${arg.activity}`);
      let buff = [];
      arg.audioStream.read(buff);
      console.log(`Buffer: ${buff}`);
    }
    connector.canceled = (sender, arg) => {
      console.log(`Cancelled, reason=${arg.reason}`);
      if (arg.reason === SpeechSDK.CancellationReason.Error) {
        console.log(`Error: code=${arg.errorCode}, details=${arg.errorDetails}`);
      }
    }
    connector.recognizing = (sender, arg) => {
      console.log(`Recognizing! in-progress text=${arg.result.text}`);
    }
    connector.recognized = (sender, arg) => {
      console.log(`Final speech-to-text result: ${arg.result.text}`);
    }
    connector.sessionStarted = (sender, arg) => {
      console.log(`Now listening! session started, id=${arg.sessionId}`);
    }
    connector.sessionStopped = (sender, arg) => {
      console.log(`Listening complete. session ended, id=${arg.sessionId}`);
    }
    setDialogueServiceConnector(connector);
    console.log("initialized");
  }

  useEffect(() => {
    initializeDialogueServiceConnector();
  }, [])

  async function playSound() {
    if (lastRecordedSound) {
      await lastRecordedSound.playAsync()
    }
  }

  async function startRecording() {
    /*
    console.log(dialogueServiceConnector)
    dialogueServiceConnector.listenOnceAsync(() => {
      console.log("Recording complete")
    });
    setRecording(true);
    */
    try {
      await Audio.requestPermissionsAsync();
      await Audio.setAudioModeAsync({
        allowsRecordingIOS: true,
        playsInSilentModeIOS: true,
      });
      const { recording } = await Audio.Recording.createAsync(
        Audio.RecordingOptionsPresets.HIGH_QUALITY
      );
      setRecording(recording);
    } catch (err) {
      console.error('Failed to start recording', err);
    }
  }

  async function stopRecording() {
    /*
    setRecording(false);
    */
    await recording.stopAndUnloadAsync();
    await Audio.setAudioModeAsync({
      allowsRecordingIOS: false,
      playsInSilentModeIOS: true,
    });
    const { sound } = await recording.createNewLoadedSoundAsync();
    setLastRecordedSound(sound)
  }

  return (
    <View style={styles.container}>
      <Button
        title={recording ? 'Stop Recording' : 'Start Recording'}
        onPress={recording ? stopRecording : startRecording}
      />
      <Button
        title='Play recording'
        onPress={playSound}
      />
    </View>
  );
}

const styles = StyleSheet.create({
  container: {
    flex: 1,
    backgroundColor: '#fff',
    alignItems: 'center',
    justifyContent: 'center',
  },
});
可行实现方案

方案一:直接使用SDK内置麦克风捕获

Azure Speech SDK的DialogServiceConnector本身支持直接从麦克风捕获音频,无需依赖expo-av录制,流程更简洁,调整后代码如下:

import 'react-native-get-random-values';
import { useState, useEffect } from 'react';
import { StyleSheet, Button, View } from 'react-native';
import * as SpeechSDK from 'microsoft-cognitiveservices-speech-sdk';

const APPLICATION_ID = '6e3b5880-918e-11ed-a04e-cf28b0773cf7';
const SPEECH_KEY = '51ecddfd91564074b3cb2cee2f9e4f52';
const SPEECH_RESOURCE_REGION = 'centralindia'

export default function App() {

  const [isListening, setIsListening] = useState(false);
  const [dialogueServiceConnector, setDialogueServiceConnector] = useState();

  function initializeDialogueServiceConnector() {
    const commandsConfig = new SpeechSDK.CustomCommandsConfig.fromSubscription(APPLICATION_ID, SPEECH_KEY, SPEECH_RESOURCE_REGION);
    const connector = new SpeechSDK.DialogServiceConnector(commandsConfig);
    
    // 监听Custom Commands返回的活动/指令
    connector.activityReceived = (sender, arg) => {
      console.log(`收到活动: ${arg.activity}`);
      // 若有返回音频,可结合expo-av处理播放(需转换格式)
      if (arg.audioStream) {
        console.log("返回音频流已接收");
      }
    };

    connector.canceled = (sender, arg) => {
      console.log(`会话取消: 原因=${arg.reason}`);
      if (arg.reason === SpeechSDK.CancellationReason.Error) {
        console.log(`错误信息: 代码=${arg.errorCode}, 详情=${arg.errorDetails}`);
      }
      setIsListening(false);
    };

    connector.recognizing = (sender, arg) => {
      console.log(`识别中: ${arg.result.text}`);
    };

    connector.recognized = (sender, arg) => {
      console.log(`最终识别结果: ${arg.result.text}`);
    };

    connector.sessionStarted = (sender, arg) => {
      console.log(`会话已启动,ID=${arg.sessionId}`);
    };

    connector.sessionStopped = (sender, arg) => {
      console.log(`会话已结束,ID=${arg.sessionId}`);
      setIsListening(false);
    };

    setDialogueServiceConnector(connector);
    console.log("DialogServiceConnector初始化完成");
  }

  useEffect(() => {
    initializeDialogueServiceConnector();
    
    // 组件卸载时清理资源
    return () => {
      if (dialogueServiceConnector) {
        dialogueServiceConnector.close();
      }
    };
  }, []);

  async function toggleListening() {
    if (!dialogueServiceConnector) return;

    if (isListening) {
      dialogueServiceConnector.stopListeningAsync();
      setIsListening(false);
    } else {
      setIsListening(true);
      // 单次监听,完成后自动停止;如需持续监听可使用listenContinuousAsync
      dialogueServiceConnector.listenOnceAsync(
        () => {
          console.log("单次监听完成");
          setIsListening(false);
        },
        (err) => {
          console.error("监听出错:", err);
          setIsListening(false);
        }
      );
    }
  }

  return (
    <View style={styles.container}>
      <Button
        title={isListening ? '停止监听' : '开始监听'}
        onPress={toggleListening}
      />
    </View>
  );
}

const styles = StyleSheet.create({
  container: {
    flex: 1,
    backgroundColor: '#fff',
    alignItems: 'center',
    justifyContent: 'center',
  },
});

方案二:使用expo-av录制的音频文件

若必须使用expo-av录制的音频,需将音频转换为16kHz采样率、16位单声道PCM格式,再通过AudioInputStream传递给SDK,核心逻辑如下(需依赖react-native-fs读取文件):

import RNFS from 'react-native-fs';

// 假设已获取expo-av录制的音频文件路径
async function sendRecordedAudioToSDK(audioFilePath) {
  if (!dialogueServiceConnector) return;

  try {
    // 读取音频文件内容
    const fileBase64 = await RNFS.readFile(audioFilePath, 'base64');
    // 转换为Buffer(需确保音频格式是16kHz 16bit mono PCM,若不是需额外转换)
    const pcmBuffer = Buffer.from(fileBase64, 'base64');

    // 创建推送式音频输入流
    const audioStream = SpeechSDK.AudioInputStream.createPushStream();
    audioStream.write(pcmBuffer);
    audioStream.close();

    // 创建音频配置并初始化连接器
    const audioConfig = SpeechSDK.AudioConfig.fromStreamInput(audioStream);
    const commandsConfig = new SpeechSDK.CustomCommandsConfig.fromSubscription(APPLICATION_ID, SPEECH_KEY, SPEECH_RESOURCE_REGION);
    const connector = new SpeechSDK.DialogServiceConnector(commandsConfig, audioConfig);

    // 绑定事件监听(同方案一)
    connector.recognized = (sender, arg) => {
      console.log(`最终识别结果: ${arg.result.text}`);
    };

    // 开始识别
    connector.listenOnceAsync(
      () => console.log("识别完成"),
      (err) => console.error("识别出错:", err)
    );
  } catch (err) {
    console.error("处理音频文件出错:", err);
  }
}

内容的提问来源于stack exchange,提问作者Aditya Azad

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.04 21:00:51