You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

pyttsx3发声时listen_in_background无回调,如何中断AI语音输出?

如何实现语音AI的实时语音中断功能?

我是论坛新手,还请多多包涵!我基于YouTube教程开发了名为Jarvis的语音AI,但始终无法在AI说话时用语音指令中断它。我尝试多种方法后改用线程处理语音播放,可即便播放在独立线程运行,listen_in_background仍无法触发回调。我该如何实现用语音指令让Jarvis中途停止说话?

最初我以为是runAndWait()的问题,但现在怀疑listen_in_background在有执行任务时无法正常工作。请问是这样吗?有没有解决办法?我希望实现更具交互性的语音中断功能。

当前代码

from os import system
import speech_recognition as sr
from playsound import playsound
from gpt4all import GPT4All
import whisper
import time
import os
import pyttsx3
import importlib
import threading

wake_word = "jarvis"
model = GPT4All("nous-hermes-llama2-13b.Q4_0.gguf", allow_download=False)
r = sr.Recognizer()
tiny_model = whisper.load_model("tiny")
base_model = whisper.load_model("base")
listening_for_wake_word = True
stop_talking = False
source = sr.Microphone()

def speak_thread(text): 
    importlib.reload(pyttsx3)
    engine = pyttsx3.init()
    engine.say(text)
    engine.runAndWait()

def speak(text):
    global stop_talking
    
    talk_thread = threading.Thread(daemon=True, target=speak_thread, name="talking", args=(text,)).start()
    
    while any("talking" in item.name for item in threading.enumerate()):
        if stop_talking:
            talk_thread.stop()
        time.sleep(1)
    
def listen_for_wake_word(audio):
    global listening_for_wake_word
    with open("wake_detect.wav", "wb") as f:
        f.write(audio.get_wav_data())
    result = tiny_model.transcribe("wake_detect.wav")
    text_input = result["text"]
    if wake_word in text_input.lower().strip():
        print("Wake word detected. Please speak your prompt to GPT4All.")
        speak("Listening")
        listening_for_wake_word = False

def prompt_gpt(audio): 
    global listening_for_wake_word
    try: 
        with open("prompt.wav", "wb") as f:
            f.write(audio.get_wav_data())
        result = base_model.transcribe("prompt.wav")
        prompt_text = result["text"]
        if len(prompt_text .strip()) == 0:
            print("I didn't catch that. Please repeat.")
            listening_for_wake_word = True
        else: 
            print("User: " + prompt_text)
            output = base_model.generate(prompt_text, max_tokens=500)
            print("GPT4All: ", output)
            speak(output)
            print("\nSay", wake_word, "to wake me up.\n")
            listening_for_wake_word = True
    except Exception as e: 
        print("Prompt error: ", e)
        
def callback (recognizer, audio): 
    global listening_for_wake_word
    global stop_talking
    
    print("Heard something")
    
    try:
        with open("temp.wav", "wb") as f:
            f.write(audio.get_wav_data())
        result = tiny_model.transcribe("wake_detect.wav")
        
        text_input = result["text"]
        
        if "stop" in text_input.lower().strip():
            print("Stopping playback...")
            stop_talking = True
            return
    except Exception as e: 
        print("Prompt error in stop part: ", e)
    
    if listening_for_wake_word: 
        listen_for_wake_word(audio)
    else: 
        prompt_gpt(audio)
    
def start_listening(): 
    with source as s: 
        r.adjust_for_ambient_noise(s, duration=2)
    print("\nSay", wake_word, "to wake me up.\n")
    r.listen_in_background(source, callback) # Why does this not send further callbacks until after audio playback?
    while True: # just to keep alive
        time.sleep(1)
        print(threading.enumerate())
        threading.get_ident()
        
if __name__ == "__main__":
    start_listening()

问题分析

  1. 线程控制错误:talk_thread.start()返回的是None,调用talk_thread.stop()会直接报错;且pyttsx3的engine实例是线程内创建的,外部无法直接控制停止。
  2. 主线程阻塞:speak函数里的while循环一直在主线程轮询,占用了主线程资源,导致listen_in_background的回调无法及时触发(speech_recognition的后台监听依赖主线程处理事件循环)。
  3. 文件读写冲突:回调里写入temp.wav,但识别时读取的是wake_detect.wav,存在文件复用导致的识别错误。
  4. 状态重置不及时:触发stop后没有重置stop_talking状态,可能影响后续播放。

修复方案

以下是调整后的完整代码,核心改动是用全局pyttsx3引擎实例实现实时中断,同时避免主线程阻塞:

from os import system
import speech_recognition as sr
from gpt4all import GPT4All
import whisper
import time
import os
import pyttsx3
import threading

wake_word = "jarvis"
model = GPT4All("nous-hermes-llama2-13b.Q4_0.gguf", allow_download=False)
r = sr.Recognizer()
tiny_model = whisper.load_model("tiny")
base_model = whisper.load_model("base")
listening_for_wake_word = True
source = sr.Microphone()

# 全局pyttsx3引擎实例,方便跨线程控制
engine = pyttsx3.init()
# 标记是否正在播放
is_talking = False

def speak_thread(text):
    global is_talking
    is_talking = True
    try:
        engine.say(text)
        engine.runAndWait()
    finally:
        is_talking = False

def speak(text):
    # 先停止当前播放(如果有的话)
    engine.stop()
    # 启动新的播放线程
    threading.Thread(daemon=True, target=speak_thread, name="talking", args=(text,)).start()

def listen_for_wake_word(audio):
    global listening_for_wake_word
    with open("wake_detect.wav", "wb") as f:
        f.write(audio.get_wav_data())
    result = tiny_model.transcribe("wake_detect.wav")
    text_input = result["text"].lower().strip()
    if wake_word in text_input:
        print("Wake word detected. Please speak your prompt to GPT4All.")
        speak("Listening")
        listening_for_wake_word = False

def prompt_gpt(audio): 
    global listening_for_wake_word
    try: 
        with open("prompt.wav", "wb") as f:
            f.write(audio.get_wav_data())
        result = base_model.transcribe("prompt.wav")
        prompt_text = result["text"].strip()
        if not prompt_text:
            print("I didn't catch that. Please repeat.")
            listening_for_wake_word = True
        else: 
            print(f"User: {prompt_text}")
            output = model.generate(prompt_text, max_tokens=500)
            print(f"GPT4All: {output}")
            speak(output)
            print(f"\nSay {wake_word} to wake me up.\n")
            listening_for_wake_word = True
    except Exception as e: 
        print(f"Prompt error: {e}")
        listening_for_wake_word = True
        
def callback(recognizer, audio): 
    global listening_for_wake_word
    
    print("Heard something")
    
    try:
        with open("temp_stop_detect.wav", "wb") as f:
            f.write(audio.get_wav_data())
        result = tiny_model.transcribe("temp_stop_detect.wav")
        text_input = result["text"].lower().strip()
        
        # 优先处理停止指令,不管当前状态
        if "stop" in text_input:
            print("Stopping playback...")
            engine.stop()
            # 如果是在对话状态,重置回唤醒词监听
            if not listening_for_wake_word:
                listening_for_wake_word = True
                print(f"\nSay {wake_word} to wake me up.\n")
            return
    except Exception as e: 
        print(f"Stop detection error: {e}")
    
    if listening_for_wake_word: 
        listen_for_wake_word(audio)
    else: 
        prompt_gpt(audio)
    
def start_listening(): 
    with source as s: 
        r.adjust_for_ambient_noise(s, duration=2)
    print(f"\nSay {wake_word} to wake me up.\n")
    # 启动后台监听,回调会在独立线程执行
    r.listen_in_background(source, callback)
    # 主线程保持存活,不要做阻塞操作
    while True:
        time.sleep(1)
        
if __name__ == "__main__":
    start_listening()

关键改动说明

  1. 全局引擎实例:创建一个全局的pyttsx3引擎,所有播放都用这个实例,这样在回调里可以直接调用engine.stop()中断播放。
  2. 移除主线程轮询:speak函数不再用while循环阻塞主线程,让listen_in_background的回调能及时处理语音输入。
  3. 优先处理停止指令:在回调里优先检测"stop"指令,不管当前处于唤醒词监听还是对话状态,都能立即中断播放并重置状态。
  4. 修复文件冲突:停止检测用单独的temp_stop_detect.wav文件,避免和唤醒词检测的文件冲突。
  5. 状态管理优化:添加is_talking标记跟踪播放状态,同时确保异常情况下也能重置listening_for_wake_word状态。

内容的提问来源于stack exchange,提问作者user23118418

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.03 16:20:25