You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用JavaScript实现浏览器麦克风音频直连HuggingFace Whisper API流式传输

实现浏览器直接传输麦克风音频到HuggingFace Whisper API(无文件/中间服务器)

你提供的代码是Node.js环境下的文件读取逻辑,浏览器中无法使用fs模块,我们可以通过浏览器原生的MediaRecorder API捕获麦克风音频,直接生成Blob对象发送到HuggingFace API,完全不需要中间服务器或本地文件存储。

完整实现步骤与代码

1. 捕获麦克风音频并生成Blob

使用MediaRecorder获取用户麦克风的音频流,实时收集音频数据块,停止录制后直接封装为Blob(二进制数据容器)。

2. 修改API请求逻辑

将原有的文件读取逻辑替换为发送Blob,fetch API原生支持将Blob作为请求体发送。

完整代码示例

<!-- 页面按钮用于触发录制 -->
<button id="start-btn">开始语音指令</button>
<button id="stop-btn">停止录制并识别</button>

<script>
// 请求麦克风权限并创建MediaRecorder实例
async function startRecording() {
  // 获取用户麦克风权限
  const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
  
  // 选择Whisper支持的音频格式,优先用flac(压缩率高),不支持则降级为wav
  let mimeType = 'audio/flac';
  if (!MediaRecorder.isTypeSupported(mimeType)) {
    console.warn('当前浏览器不支持FLAC格式,将使用WAV格式');
    mimeType = 'audio/wav';
  }

  const recorder = new MediaRecorder(stream, { mimeType });
  const audioChunks = [];

  // 收集音频数据块
  recorder.addEventListener('dataavailable', (event) => {
    if (event.data.size > 0) {
      audioChunks.push(event.data);
    }
  });

  // 录制停止后,发送音频到Whisper API
  recorder.addEventListener('stop', async () => {
    const audioBlob = new Blob(audioChunks, { type: mimeType });
    try {
      const transcription = await queryWhisper(audioBlob);
      console.log('语音识别结果:', transcription);
      // 这里添加你的邮件控制逻辑,比如根据识别文本执行发送/归档等操作
    } catch (err) {
      console.error('识别失败:', err);
    }
    // 停止麦克风流释放资源
    stream.getTracks().forEach(track => track.stop());
  });

  return recorder;
}

// 发送音频Blob到HuggingFace Whisper API
async function queryWhisper(audioBlob) {
  const response = await fetch(
    "https://api-inference.huggingface.co/models/openai/whisper-medium",
    {
      headers: { 
        Authorization: "Bearer YOUR_HUGGINGFACE_TOKEN", // 替换为你的API Token
        "Content-Type": audioBlob.type // 匹配音频格式的Content-Type
      },
      method: "POST",
      body: audioBlob, // 直接传入Blob作为请求体
    }
  );

  if (!response.ok) {
    throw new Error(`API请求失败: ${response.status} ${response.statusText}`);
  }

  const result = await response.json();
  return result.text; // 返回识别出的文本内容
}

// 绑定按钮事件
let recorderInstance;
document.getElementById('start-btn').addEventListener('click', async () => {
  recorderInstance = await startRecording();
  recorderInstance.start();
  console.log('正在录制,请说话...');
});

document.getElementById('stop-btn').addEventListener('click', () => {
  if (recorderInstance && recorderInstance.state !== 'inactive') {
    recorderInstance.stop();
    console.log('录制停止,正在识别...');
  }
});
</script>

关键注意事项

  • 权限要求:浏览器会强制弹出麦克风授权窗口,用户必须同意才能捕获音频,无法绕过。
  • 格式兼容性:不同浏览器对音频格式支持有差异,代码中做了自动降级处理,确保兼容性。
  • API Token:替换代码中的YOUR_HUGGINGFACE_TOKEN为你在HuggingFace平台获取的API密钥。
  • 错误处理:代码中包含基础错误捕获,实际项目中可以扩展为用户友好的提示(比如网络失败、识别超时等)。

内容的提问来源于stack exchange,提问作者Aadil Sayad

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.30 14:38:12