You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python多进程共享变量问题求助:语音控制鼠标失效

语音控制鼠标脚本多进程模式失效问题求助

我写了个语音控制鼠标的脚本,功能包括录音、生成频谱图、调用训练好的模型预测指令,进而控制鼠标移动。脚本用Threading模式能正常跑,但切到Multiprocessing模式就失效了。我推测是因为move_mouse()跑在独立进程里,导致is_moving和movement_direction这两个变量没法跨进程更新。我对多进程的共享变量机制不太懂,求技术帮助。

当前代码

import pyautogui
from tensorflow import keras
import numpy as np
from matplotlib import pyplot as plt
from sys import byteorder
import librosa
import librosa.display
import tensorflow as tf
from array import array
import numpy as np
import pyaudio
import sys
from keras.models import load_model
from PIL import Image
import threading
import matplotlib
matplotlib.use('Agg')
import multiprocessing

model = keras.models.load_model(r'ModeloFinal.h5')

MOVEMENT_SPEED = 10

np.set_printoptions(threshold=sys.maxsize)
img_height = int(480/2) #432 #28
img_width = int(640/2)


class AudioRecorder:
    def __init__(self):
        self.rate = 44100
        self.threshold = 0.018 # silence threshold {Need to experiment with it}
        self.chunk_size = 1024
        self.format = pyaudio.paFloat32
        self._pyaudio = pyaudio.PyAudio()

    def isSilent(self, data):
        # Returns true if below the silence threshold
        return max(data) < self.threshold

    def trim(self, data):
        # Trim the blanks at the start and end
        def _trim(data):
            started = False
            r = array('f')
            for i in data:
                if not started and abs(i) > self.threshold:
                    started = True
                    r.append(i)
                elif started:
                    r.append(i)
            return r

        # First trim the left side
        data = _trim(data)
        data.reverse()

        # Then trim the right side
        data = _trim(data)
        data.reverse()

        return data

    def addSilence(self, data):
        # Adds silence to the start and end of 0.1 seconds
        r = array('f', [0 for i in range(int(0.1 * self.rate))])
        r.extend(data)
        r.extend([0 for i in range(int(0.1 * self.rate))])

        return r

    def record(self):
        stream = self._pyaudio.open(format=self.format, channels=1, rate=self.rate, input=True, output=True, frames_per_buffer=self.chunk_size)
        numSilent = 0
        started = False
        r = array('f')

        while True:
            # Has to be little endian and signed short
            data = array('f', stream.read(self.chunk_size))

            if byteorder == 'big':
                data.byteswap()

            r.extend(data)
            silent = self.isSilent(data)

            if silent and started:
                numSilent += 1
            elif not silent and not started:
                started = True
                print('Noise detected, recording beginning')
            if started and numSilent > 20:  # If there are 30 silences, break (doesn't handle long sentences).
                break

        stream.stop_stream()
        stream.close()

        r = self.trim(r)
        r = self.addSilence(r)

        r = np.array(r, dtype=np.float32)

        fig, ax = plt.subplots()

        melspec = librosa.feature.melspectrogram(y=r, sr=self.rate, n_mels=256, fmax=22000, hop_length=8)
        norm_melspec = librosa.core.power_to_db(melspec, ref=np.max)
        img = librosa.display.specshow(norm_melspec, x_axis='time', y_axis='mel', sr=self.rate, fmax=22000, ax=ax)
        plt.savefig(('spokenWord.png'), transparent=True)
        plt.close()

        imgToPredict = Image.open('spokenWord.png').convert('RGB')
        imgToPredict = imgToPredict.resize((img_width, img_height))
        im_np = np.asarray(imgToPredict)
        print(np.shape(im_np))
        numpydata = im_np.reshape(1, img_height, img_width, 3)
        predictions = model.predict(numpydata)
        pred_labels = np.argmax(predictions, axis=1)
        print("Result")
        print(pred_labels)

        if pred_labels[0] == 0:
            mouse_controller.move_up()
        elif pred_labels[0] == 1:
            mouse_controller.move_down()
        elif pred_labels[0] == 2:
            mouse_controller.move_left()
        elif pred_labels[0] == 3:
            mouse_controller.move_right()
        elif pred_labels[0] == 4:
            mouse_controller.pause_movement()
        else:
            mouse_controller.click_mouse()

        print('Returning to listening')

        if silent:
            self.record()

    def listen(self):
        print('Listening beginning')
        self.record()


class MouseController:
    is_moving = False
    movement_direction = None

    def move_mouse(self):
        if self.is_moving:
            print('abc')
            if self.movement_direction == 'up':
                pyautogui.move(0, -MOVEMENT_SPEED)
            elif self.movement_direction == 'down':
                pyautogui.move(0, MOVEMENT_SPEED)
            elif self.movement_direction == 'left':
                pyautogui.move(-MOVEMENT_SPEED, 0)
            elif self.movement_direction == 'right':
                pyautogui.move(MOVEMENT_SPEED, 0)

    def move_up(self):
        print('UP')
        if not self.is_moving:
            self.is_moving = True
            self.movement_direction = 'up'

    def move_down(self):
        print('DOWN')
        if not self.is_moving:
            self.is_moving = True
            self.movement_direction = 'down'

    def move_left(self):
        print('LEFT')
        if not self.is_moving:
            self.is_moving = True
            self.movement_direction = 'left'

    def move_right(self):
        print('RIGHT')
        if not self.is_moving:
            self.is_moving = True
            self.movement_direction = 'right'

    def click_mouse(self):
        self.is_moving = False
        pyautogui.click()

    def pause_movement(self):
        self.is_moving = False


a = AudioRecorder()
mouse_controller = MouseController()


def audio_thread():
    a.listen()


def mouse_thread():
    while True:
        mouse_controller.move_mouse()


if __name__ == '__main__':
    audio_process = multiprocessing.Process(target=audio_thread)
    mouse_process = multiprocessing.Process(target=mouse_thread)

    audio_process.start()
    mouse_process.start()

    audio_process.join()
    mouse_process.join()

问题根源

多进程模式下,每个进程会复制一份父进程的内存空间,音频进程里修改的mouse_controller实例,和鼠标进程里的是完全独立的两个对象,互相看不到对方的修改。这就是线程模式能用(线程共享同一份内存)但多进程不行的核心原因。

修复方案

方案1:使用多进程共享变量

把MouseController里的普通变量换成multiprocessing提供的共享对象,同时加锁避免竞态条件:

class MouseController:
    def __init__(self):
        # 布尔型共享变量,'b'代表字节型(对应布尔)
        self.is_moving = multiprocessing.Value('b', False)
        # 字节数组存方向字符串,长度设为足够存方向的长度
        self.movement_direction = multiprocessing.Array('c', 10)

    def move_mouse(self):
        # 加锁保证读写原子性
        with self.is_moving.get_lock():
            if self.is_moving.value:
                direction = self.movement_direction.value.decode('utf-8')
                if direction == 'up':
                    pyautogui.move(0, -MOVEMENT_SPEED)
                elif direction == 'down':
                    pyautogui.move(0, MOVEMENT_SPEED)
                elif direction == 'left':
                    pyautogui.move(-MOVEMENT_SPEED, 0)
                elif direction == 'right':
                    pyautogui.move(MOVEMENT_SPEED, 0)

    def move_up(self):
        print('UP')
        with self.is_moving.get_lock():
            if not self.is_moving.value:
                self.is_moving.value = True
                self.movement_direction.value = b'up'

    def move_down(self):
        print('DOWN')
        with self.is_moving.get_lock():
            if not self.is_moving.value:
                self.is_moving.value = True
                self.movement_direction.value = b'down'

    def move_left(self):
        print('LEFT')
        with self.is_moving.get_lock():
            if not self.is_moving.value:
                self.is_moving.value = True
                self.movement_direction.value = b'left'

    def move_right(self):
        print('RIGHT')
        with self.is_moving.get_lock():
            if not self.is_moving.value:
                self.is_moving.value = True
                self.movement_direction.value = b'right'

    def click_mouse(self):
        with self.is_moving.get_lock():
            self.is_moving.value = False
        pyautogui.click()

    def pause_movement(self):
        with self.is_moving.get_lock():
            self.is_moving.value = False

方案2:使用进程间通信队列

用multiprocessing.Queue让音频进程发送指令,鼠标进程接收并执行,天然避免竞态问题:

# 全局指令队列
command_queue = multiprocessing.Queue()

class AudioRecorder:
    # 其他方法不变,修改record里的指令调用部分
    def record(self):
        # ... 预测逻辑不变 ...
        if pred_labels[0] == 0:
            command_queue.put('up')
        elif pred_labels[0] == 1:
            command_queue.put('down')
        elif pred_labels[0] == 2:
            command_queue.put('left')
        elif pred_labels[0] == 3:
            command_queue.put('right')
        elif pred_labels[0] == 4:
            command_queue.put('pause')
        else:
            command_queue.put('click')
        # ... 其他逻辑不变 ...

class MouseController:
    def __init__(self):
        self.is_moving = False
        self.movement_direction = None

    def run(self, queue):
        while True:
            # 先处理队列里的所有待执行指令
            while not queue.empty():
                command = queue.get()
                if command == 'up':
                    print('UP')
                    if not self.is_moving:
                        self.is_moving = True
                        self.movement_direction = 'up'
                elif command == 'down':
                    print('DOWN')
                    if not self.is_moving:
                        self.is_moving = True
                        self.movement_direction = 'down'
                elif command == 'left':
                    print('LEFT')
                    if not self.is_moving:
                        self.is_moving = True
                        self.movement_direction = 'left'
                elif command == 'right':
                    print('RIGHT')
                    if not self.is_moving:
                        self.is_moving = True
                        self.movement_direction = 'right'
                elif command == 'pause':
                    self.is_moving = False
                elif command == 'click':
                    self.is_moving = False
                    pyautogui.click()
            # 执行鼠标移动
            if self.is_moving:
                if self.movement_direction == 'up':
                    pyautogui.move(0, -MOVEMENT_SPEED)
                elif self.movement_direction == 'down':
                    pyautogui.move(0, MOVEMENT_SPEED)
                elif self.movement_direction == 'left':
                    pyautogui.move(-MOVEMENT_SPEED, 0)
                elif self.movement_direction == 'right':
                    pyautogui.move(MOVEMENT_SPEED, 0)

# 修改进程启动逻辑
if __name__ == '__main__':
    a = AudioRecorder()
    mouse_controller = MouseController()
    
    audio_process = multiprocessing.Process(target=audio_thread)
    # 把队列传给鼠标进程
    mouse_process = multiprocessing.Process(target=mouse_controller.run, args=(command_queue,))

    audio_process.start()
    mouse_process.start()

    audio_process.join()
    mouse_process.join()

方案选择

  • 若仅需共享少量状态,方案1(共享变量)更直接,但必须注意加锁
  • 若需要传递复杂指令或扩展功能,方案2(队列)更灵活,无需手动处理锁

内容的提问来源于stack exchange,提问作者phpc

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.19 00:29:54