You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在PyQt类外部更新setTitle中的语音识别文本

解决PyQt实时波形窗口中更新语音识别文本的问题

问题根源

  • 多进程内存隔离:GUI进程与录音/识别进程拥有独立内存空间,全局变量phrase无法跨进程同步更新。
  • 局部变量覆盖:Speech_Recog函数内定义的phrase是局部变量,未修改全局作用域的同名变量。
  • 流程阻塞:主线程启动GUI进程后直接执行录音、识别逻辑,执行完成后立即退出,导致GUI进程无法持续接收更新。

解决方案

使用multiprocessing.Queue实现跨进程通信,将录音与识别逻辑放到子进程中执行,GUI进程通过定时器读取队列中的识别结果并更新标题。

修改后的完整代码

'''GUI'''
import struct
from PyQt5 import QtWidgets
from PyQt5.QtWidgets import QApplication
import sys

'''Graph'''
import pyqtgraph as pg
from PyQt5 import QtCore
import numpy as np

'''Audio Processing'''
import pyaudio
import wave
import speech_recognition as sr
import multiprocessing as mlti

FORMAT = pyaudio.paInt16
CHANNELS = 1
RATE = 44100
CHUNK = 1024 * 2
seconds = 6

class MainWindow(QtWidgets.QMainWindow):
    def __init__(self, msg_queue, *args, **kwargs):
        super(MainWindow, self).__init__(*args, **kwargs)
        pg.setConfigOptions(antialias=True)
        self.traces = dict()
        self.msg_queue = msg_queue  # 接收跨进程通信队列
        self.current_phrase = "..."

        '''Display'''
        self.graphWidget = pg.PlotWidget()
        self.setCentralWidget(self.graphWidget)
        self.setWindowTitle("Waveform")
        self.setGeometry(55, 115, 970, 449)

        '''Data'''
        self.x = np.arange(0, 2 * CHUNK, 2)
        self.f = np.linspace(0, RATE // 2, CHUNK // 2)

        '''Animate'''
        self.timer = QtCore.QTimer()
        self.timer.setInterval(50)
        self.timer.timeout.connect(self.update)
        self.timer.start()  

    def set_plotdata(self, name, data_x, data_y):
        if name in self.traces:
            self.traces[name].setData(data_x, data_y)
        else:
            if name == 'waveform':
                self.traces[name] = self.graphWidget.plot(pen='c', width=3)
                self.graphWidget.setYRange(0, 255, padding=0)
                self.graphWidget.setXRange(0, 2 * CHUNK, padding=0.005)

    def update(self):
        # 读取队列中的识别结果
        while not self.msg_queue.empty():
            self.current_phrase = self.msg_queue.get()
            # 更新窗口标题(或graphWidget标题)
            self.setWindowTitle(f"Waveform - {self.current_phrase}")
            self.graphWidget.setTitle(self.current_phrase, color="w", size="30pt")

        # 实时波形绘制
        p = pyaudio.PyAudio()
        stream = p.open(
            format=FORMAT,
            channels=CHANNELS,
            rate=RATE,
            input=True,
            output=True,
            frames_per_buffer=CHUNK,
        )
        self.wf_data = stream.read(CHUNK)
        self.wf_data = struct.unpack(str(2 * CHUNK) + 'B', self.wf_data)
        self.wf_data = np.array(self.wf_data, dtype='b')[::2] + 128
        self.set_plotdata(name='waveform', data_x=self.x, data_y=self.wf_data)
        stream.stop_stream()
        stream.close()
        p.terminate()

def main(msg_queue):
    app = QtWidgets.QApplication(sys.argv)
    win = MainWindow(msg_queue)
    win.show()
    sys.exit(app.exec_())

def Record():
    frames = []
    p = pyaudio.PyAudio()
    stream = p.open(
        format=FORMAT,
        channels=CHANNELS,
        rate=RATE,
        input=True,
        output=True,
        frames_per_buffer=CHUNK,
    )
    for i in range(0, int(RATE/CHUNK*seconds)):
        data = stream.read(CHUNK)
        frames.append(data)
    stream.stop_stream()
    stream.close()
    p.terminate()

    # 保存录音文件
    obj = wave.open("output.wav", "wb")
    obj.setnchannels(CHANNELS)
    obj.setsampwidth(p.get_sample_size(FORMAT))
    obj.setframerate(RATE)
    obj.writeframes(b"".join(frames))
    obj.close()

def Speech_Recog(msg_queue):
    print("Function Started")
    r = sr.Recognizer()
    with sr.AudioFile("output.wav") as source:
        r.adjust_for_ambient_noise(source, duration=1)
        audio = r.listen(source)
        phrase = ""
        try:
            phrase = r.recognize_google(audio, language='pt-BR')
            print(phrase)
        except sr.UnknownValueError:
            phrase = "Not understood"
            print(phrase)
        # 将识别结果放入队列
        msg_queue.put(phrase)

def audio_process(msg_queue):
    Record()
    Speech_Recog(msg_queue)

if __name__ == '__main__':
    # 创建跨进程通信队列
    msg_queue = mlti.Queue()
    # 启动GUI进程
    p1 = mlti.Process(target=main, args=(msg_queue,))
    p1.start()
    # 启动录音识别进程
    p2 = mlti.Process(target=audio_process, args=(msg_queue,))
    p2.start()
    # 等待进程结束
    p2.join()
    p1.join()

关键修改说明

  1. 跨进程通信队列:创建multiprocessing.Queue,作为GUI进程与录音识别进程之间的消息传递通道。
  2. GUI初始化接收队列:MainWindow初始化时接收队列,在定时器更新函数中读取队列内容,更新窗口标题和波形图标题。
  3. 封装录音识别流程:将Record和Speech_Recog封装到audio_process函数中,作为独立子进程执行,避免阻塞GUI主线程。
  4. 避免全局资源冲突:将PyAudio的stream创建/销毁放到局部作用域,避免多进程共享全局stream导致的资源冲突。

内容的提问来源于stack exchange,提问作者Gabriel _pcm

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.17 22:20:33