You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python多进程启动报错:TypeError: cannot pickle 'weakref' object

TypeError: cannot pickle 'weakref' object 问题排查与解决

问题描述

尝试用multiprocessing加速程序运行时触发TypeError: cannot pickle 'weakref' object错误,相同方法在另一程序中可正常运行,已尝试相关方案但未解决。

问题代码

import multiprocessing
from scipy import stats
import numpy as np
import pandas as pd
class T_TestFeature:
    def __init__(self, data, classes):
        self.data = data
        self.classes = classes 
        self.manager = multiprocessing.Manager()
        self.pval = self.manager.list()
        
    def preform(self):
        process = []
        for i in range(10):
            process.append(multiprocessing.Process(target=self.t_test, args=(i,)))

        for p in process:
            p.start()

        for p in process:
            p.join()

    def t_test(self, k):
        index_samples = np.array(self.data)[:,k]
        rs1 = [index_samples[i] for i in range(len(index_samples)) if self.classes[i] == "Virginia"]
        rs2 = [index_samples[i] for i in range(len(index_samples)) if self.classes[i] != "Virginia"]
        self.pval.append(stats.ttest_ind(rs1, rs2, equal_var=False).pvalue)

def main():
    df = pd.read_excel("/Users/xxx/Documents/Project/src/flattened.xlsx")
    flattened = df.values.T
    y = df.columns
    result = T_TestFeature(flattened, y)
    result.preform()
    print(result.pval)

if __name__ == "__main__":
    main()

报错回溯

Traceback (most recent call last):
  File "/Users/xxx/Documents/Project/src/t_test.py", line 41, in <module>
    main()
  File "/Users/xxx/Documents/Project/src/t_test.py", line 37, in main
    result.preform()
  File "/Users/xxx/Documents/Project/src/t_test.py", line 21, in preform
    p.start()
  File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/process.py", line 121, in start
    self._popen = self._Popen(self)
  File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/context.py", line 284, in _Popen
    return Popen(process_obj)
  File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/popen_spawn_posix.py", line 32, in __init__
    super().__init__(process_obj)
  File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/popen_fork.py", line 19, in __init__
    self._launch(process_obj)
  File "/Users/x/opt/anaconda3/lib/python3.9/multiprocessing/popen_spawn_posix.py", line 47, in _xxlaunch
    reduction.dump(process_obj, fp)
  File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/reduction.py", line 60, in dump
    ForkingPickler(file, protocol).dump(obj)
TypeError: cannot pickle 'weakref' object

问题原因

将类方法self.t_test作为multiprocessing.Process的target时,整个类实例会被序列化(pickle)传递给子进程。类实例中包含的multiprocessing.Manager()对象内部存在无法被pickle序列化的弱引用结构;同时df.columns返回的pd.Index对象也可能包含弱引用,双重因素导致序列化失败。

解决方案

方案1:使用进程池替代手动创建Process,分离任务逻辑

把t_test逻辑改为独立函数,通过进程池批量执行任务,用返回值收集结果,避免传递整个类实例:

import multiprocessing
from scipy import stats
import numpy as np
import pandas as pd

def t_test(args):
    data, classes, k = args
    index_samples = data[:, k]
    # 用numpy布尔索引替代列表推导,提升效率
    mask = classes == "Virginia"
    rs1 = index_samples[mask]
    rs2 = index_samples[~mask]
    return stats.ttest_ind(rs1, rs2, equal_var=False).pvalue

def main():
    df = pd.read_excel("/Users/xxx/Documents/Project/src/flattened.xlsx")
    flattened = df.values.T
    # 将pd.Index转为普通列表,避免序列化问题
    y = df.columns.tolist()
    
    # 用进程池自动管理子进程
    with multiprocessing.Pool() as pool:
        tasks = [(flattened, y, k) for k in range(10)]
        pval = pool.map(t_test, tasks)
    
    print(pval)

if __name__ == "__main__":
    main()

方案2:保留类结构,解耦序列化依赖

通过静态方法+队列传递结果,避免序列化整个类实例,只传递必要的可序列化数据:

import multiprocessing
from scipy import stats
import numpy as np
import pandas as pd

class T_TestFeature:
    def __init__(self, data, classes):
        self.data = data
        self.classes = classes 
        self.pval = []
        
    def preform(self):
        with multiprocessing.Manager() as manager:
            result_queue = manager.Queue()
            processes = []
            for i in range(10):
                p = multiprocessing.Process(
                    target=self._t_test_task,
                    args=(self.data, self.classes, i, result_queue)
                )
                processes.append(p)
                p.start()
            
            for p in processes:
                p.join()
            
            # 从队列收集结果
            while not result_queue.empty():
                self.pval.append(result_queue.get())
    
    @staticmethod
    def _t_test_task(data, classes, k, result_queue):
        index_samples = data[:, k]
        mask = classes == "Virginia"
        rs1 = index_samples[mask]
        rs2 = index_samples[~mask]
        result_queue.put(stats.ttest_ind(rs1, rs2, equal_var=False).pvalue)

def main():
    df = pd.read_excel("/Users/xxx/Documents/Project/src/flattened.xlsx")
    flattened = df.values.T
    y = df.columns.tolist()
    result = T_TestFeature(flattened, y)
    result.preform()
    print(result.pval)

if __name__ == "__main__":
    main()

关键优化点

  • 将pd.Index转为普通列表(.tolist()),规避索引对象的序列化问题
  • 用numpy布尔索引替代列表推导,提升数据筛选效率
  • 通过进程池或队列传递结果,避免共享类实例中的Manager对象,减少序列化依赖

内容的提问来源于stack exchange,提问作者Grid Varavithya Vitvara

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.24 22:39:20