You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Flutter speech-to-text插件监听状态下自动停止音频捕获问题求助

问题描述

我在Flutter中使用speech-to-text组件实现了语音转文本服务,功能为监听用户语音,当用户说出“period”时将捕获的句子发送至Flask后端,代码如下:

class _AudioInputViewState extends State<AudioInputView> {
  late UnityWidgetController _unityWidgetController;
  String sentence1 ="";
  String sentence2 ="";
  List<String> sentences=[];
  List<String> sentenceTokens=[];
  int starterIndex=0;
  final Map<String, HighlightedWord> _highlights = {
    'period': HighlightedWord(
      onTap: () => print('period'),
      textStyle: const TextStyle(
        color: Colors.yellow,
        fontWeight: FontWeight.bold,
      ),
    ),
  };
  late stt.SpeechToText _speech;
  bool _isListening = false;
  String _text = 'Press the button and start speaking';
  double _confidence = 1.0;

  @override
  void initState() {
    super.initState();
    _speech = stt.SpeechToText();
  }

  @override
  Widget build(BuildContext context) {
    return Scaffold(
      appBar: AppBar(

      ),
      body:
      Container(
        decoration: BoxDecoration(
          gradient: LinearGradient(
            begin: Alignment.topLeft,
            end: Alignment.bottomRight,
            colors: [Colors.blue.shade700, Colors.green.shade700],
          ),
        ),
        child: SafeArea(
          child: Column(
            mainAxisAlignment: MainAxisAlignment.center,
            children: [
              Text(
                "Translator App",
                style: TextStyle(
                  fontSize: 28,
                  fontWeight: FontWeight.bold,
                  color: Colors.white,
                ),
              ),
              SizedBox(height: 30),
              SingleChildScrollView(
                reverse: true,
                child: Container(
                  padding: const EdgeInsets.fromLTRB(30.0, 30.0, 30.0, 150.0),
                  child: TextHighlight(
                    text: _text,
                    words: _highlights,
                    textStyle: const TextStyle(
                      fontSize: 32.0,
                      color: Colors.black,
                      fontWeight: FontWeight.w400,
                    ),
                  ),
                ),
              ),
              SizedBox(height: 30),
              AvatarGlow(
                animate: _isListening,
                glowColor: Theme
                    .of(context)
                    .primaryColor,
                endRadius: 75.0,
                duration: const Duration(milliseconds: 500),
                repeatPauseDuration: Duration(milliseconds: 500),
                repeat: true,
                child: FloatingActionButton(
                  onPressed: _listen,
                  child: Icon(
                    _isListening ? Icons.mic : Icons.mic_none,
                    size: 32,
                  ),
                  backgroundColor: Colors.white,
                  foregroundColor: Colors.black,
                ),
              ),
            ],
          ),
        ),
      ),
    );
  }
  void _listen() async {
    final TranslatorProvider translatorProvider =
    Provider.of<TranslatorProvider>(context, listen: false);
    if (!_isListening) {
      bool available = await _speech.initialize(
        onStatus: (val) => print('onStatus: $val'),
        onError: (val) => print('onError: $val'),
      );
      if (available) {
        setState(() => _isListening = true);
        _speech.listen(
          onResult: (val) {
            setState(() {
              _text = val.recognizedWords;
              sentenceTokens.clear();
              List<String> words = _text.split(" ");
              for (int i = starterIndex; i < words.length; i++) {
                if (words[i] == "period") {
                  String sentence = sentenceTokens.join(' ');
                  sentences.add(sentence);
                  sentenceTokens.clear();
                  print(sentences);
                  print("---------------------------");
                  print(sentences.last);
                  starterIndex = i + 1;
                  translatorProvider.sendText(sentences.last).then((_) {
                    logger.d("TEXT SENT : ${sentences.last}");
                  }).catchError((e) {
                    logger.e(e);
                  });
                  break;
                } else {
                  sentenceTokens.add(words[i]);
                }
              }
              if (val.hasConfidenceRating && val.confidence > 0) {
                _confidence = val.confidence;
              }
            });
          },
          partialResults: true, // enable partial results
        );
      }
    } else {
      setState(() => _isListening = false);
      _speech.stop();
      print('Listened text: $_text');
    }
  }
}

目前遇到的问题是:即使处于_isListening监听状态,短暂停顿或静音后插件会自动停止捕获音频。请问这属于该插件的正常现象吗?是否有办法实现持续监听音频?


解答

1. 这是插件的正常现象吗?

是的,这属于speech-to-text插件的默认行为。该插件依赖设备系统的语音识别服务(如Android的Google语音识别、iOS的Siri语音识别),这类系统服务在检测到短暂静音或停顿后,会自动结束当前识别会话,进而导致插件停止监听。

2. 如何实现持续监听音频?

可以通过以下两种方式解决:

方式一:利用onStatus回调重启监听

在初始化_speech.initialize时,通过onStatus回调监听识别状态变化,当检测到会话结束(状态变为notListening)且仍需持续监听时,重新调用_speech.listen:

bool available = await _speech.initialize(
  onStatus: (val) {
    print('onStatus: $val');
    // 当状态变为未监听,且需要持续监听时,重新启动会话
    if (val == 'notListening' && _isListening) {
      _speech.listen(
        onResult: (val) {
          // 保留原有的结果处理逻辑
          setState(() {
            _text = val.recognizedWords;
            sentenceTokens.clear();
            List<String> words = _text.split(" ");
            for (int i = starterIndex; i < words.length; i++) {
              if (words[i] == "period") {
                String sentence = sentenceTokens.join(' ');
                sentences.add(sentence);
                sentenceTokens.clear();
                print(sentences);
                print("---------------------------");
                print(sentences.last);
                starterIndex = i + 1;
                translatorProvider.sendText(sentences.last).then((_) {
                  logger.d("TEXT SENT : ${sentences.last}");
                }).catchError((e) {
                  logger.e(e);
                });
                break;
              } else {
                sentenceTokens.add(words[i]);
              }
            }
            if (val.hasConfidenceRating && val.confidence > 0) {
              _confidence = val.confidence;
            }
          });
        },
        partialResults: true,
      );
    }
  },
  onError: (val) => print('onError: $val'),
);

方式二:调整监听参数延长静音超时

speech-to-text的listen方法提供了listenFor和pauseFor参数,可设置最长监听时长和静音超时时长。增大pauseFor的值,能让系统在更长时间的静音后才结束会话:

_speech.listen(
  onResult: (val) {
    // 保留原有的结果处理逻辑
  },
  partialResults: true,
  listenFor: Duration(hours: 1), // 设置较长的总监听时长
  pauseFor: Duration(seconds: 5), // 设置静音5秒后才结束会话
);

注意:pauseFor的最大值受系统语音识别服务限制,部分系统可能不支持过长的静音超时设置,此时需结合方式一的回调重启监听,才能稳定实现持续监听。


内容的提问来源于stack exchange,提问作者Kavishka Rajapakshe

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.28 19:00:40