You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

移除startSpeakTextAsyncButton后Azure TTS无语音输出求助

问题分析与解决

核心问题原因

  1. 浏览器自动播放安全限制
    现代浏览器禁止页面加载阶段(如DOMContentLoaded事件中)自动播放音频,必须由用户主动交互(点击、触摸等操作)触发音频播放。原代码通过按钮点击触发,符合浏览器的交互要求;改写后直接在页面加载时执行合成,即使音频合成完成,浏览器也会阻止播放。

  2. 代码执行顺序错误
    改写后的代码中,创建语音配置和合成器的逻辑在SpeechSDK变量赋值之前执行:

// 先执行了合成逻辑,此时SpeechSDK未定义
var speechConfig = SpeechSDK.SpeechConfig.fromSubscription(...)
// 之后才赋值SpeechSDK
if (!!window.SpeechSDK) {
    SpeechSDK = window.SpeechSDK;
}

虽然你反馈返回了"synthesis finished",但实际运行中这里应该会抛出ReferenceError,除非SpeechSDK已全局挂载。但顺序错误会导致合成器初始化异常,间接影响音频播放。

修正后的代码示例

方案1:保留轻量交互触发(符合浏览器规则)

如果需要页面加载后自动触发,可增加一个用户确认步骤,比如页面显示"点击开始播放"的提示,用户点击后执行合成:

var subscriptionKey, serviceRegion;
var phraseDiv, resultDiv;
var SpeechSDK;
var synthesizer;

document.addEventListener("DOMContentLoaded", function () {
    subscriptionKey = document.querySelector("#subscriptionKey");
    serviceRegion = document.querySelector("#serviceRegion");
    phraseDiv = document.querySelector("#phraseDiv");
    resultDiv = document.querySelector("#resultDiv");

    // 先确认SpeechSDK加载完成
    if (!!window.SpeechSDK) {
        SpeechSDK = window.SpeechSDK;
        document.querySelector('#content').style.display = 'block';
        document.querySelector('#warning').style.display = 'none';

        if (typeof RequestAuthorizationToken === "function") {
            RequestAuthorizationToken();
        }

        // 创建触发元素,替代原按钮
        var triggerBtn = document.createElement('button');
        triggerBtn.innerText = '开始播放语音';
        document.body.appendChild(triggerBtn);
        
        triggerBtn.addEventListener('click', function() {
            triggerBtn.disabled = true;
            runTTS();
        });
    } else {
        document.querySelector('#warning').style.display = 'block';
        return;
    }
});

function runTTS() {
    if (subscriptionKey.value === "" || subscriptionKey.value === "subscription") {
        alert("请输入你的微软认知服务语音订阅密钥!");
        return;
    }

    var speechConfig = SpeechSDK.SpeechConfig.fromSubscription(subscriptionKey.value, serviceRegion.value);
    speechConfig.speechSynthesisVoiceName = "en-US-JennyNeural";

    synthesizer = new SpeechSDK.SpeechSynthesizer(speechConfig);
    let inputText = phraseDiv.value;

    synthesizer.speakTextAsync(
        inputText,
        function (result) {
            if (result.reason === SpeechSDK.ResultReason.SynthesizingAudioCompleted) {
                resultDiv.innerHTML += "合成完成:[" + inputText + "]\n";
            } else if (result.reason === SpeechSDK.ResultReason.Canceled) {
                resultDiv.innerHTML += "合成失败,错误详情:" + result.errorDetails + "\n";
            }
            console.log(result);
            synthesizer.close();
            synthesizer = undefined;
        },
        function (err) {
            resultDiv.innerHTML += "错误:" + err + "\n";
            console.log(err);
            synthesizer.close();
            synthesizer = undefined;
        }
    );
}

方案2:结合PHP传递文本(后端传参给前端)

如果需要PHP动态传递要合成的文本,可以在PHP页面中输出JS变量:

<?php
// PHP后端定义要合成的文本
$ttsText = "Hello from PHP integrated TTS";
?>

<!-- 前端部分 -->
<script>
// 接收PHP传递的文本
var inputText = "<?php echo $ttsText; ?>";

// 其余JS逻辑同方案1,在用户交互后使用inputText进行合成
</script>

关键注意事项

  • 必须遵守浏览器的自动播放政策,所有音频播放必须由用户主动交互触发,无法完全无交互自动播放。
  • 确保SpeechSDK加载完成后再执行合成逻辑,避免因未初始化导致的异常。

内容的提问来源于stack exchange,提问作者phs

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.04 07:35:20