Python多进程启动报错:TypeError: cannot pickle 'weakref' object
TypeError: cannot pickle 'weakref' object 问题排查与解决
问题描述
尝试用multiprocessing加速程序运行时触发TypeError: cannot pickle 'weakref' object错误,相同方法在另一程序中可正常运行,已尝试相关方案但未解决。
问题代码
import multiprocessing from scipy import stats import numpy as np import pandas as pd class T_TestFeature: def __init__(self, data, classes): self.data = data self.classes = classes self.manager = multiprocessing.Manager() self.pval = self.manager.list() def preform(self): process = [] for i in range(10): process.append(multiprocessing.Process(target=self.t_test, args=(i,))) for p in process: p.start() for p in process: p.join() def t_test(self, k): index_samples = np.array(self.data)[:,k] rs1 = [index_samples[i] for i in range(len(index_samples)) if self.classes[i] == "Virginia"] rs2 = [index_samples[i] for i in range(len(index_samples)) if self.classes[i] != "Virginia"] self.pval.append(stats.ttest_ind(rs1, rs2, equal_var=False).pvalue) def main(): df = pd.read_excel("/Users/xxx/Documents/Project/src/flattened.xlsx") flattened = df.values.T y = df.columns result = T_TestFeature(flattened, y) result.preform() print(result.pval) if __name__ == "__main__": main()
报错回溯
Traceback (most recent call last): File "/Users/xxx/Documents/Project/src/t_test.py", line 41, in <module> main() File "/Users/xxx/Documents/Project/src/t_test.py", line 37, in main result.preform() File "/Users/xxx/Documents/Project/src/t_test.py", line 21, in preform p.start() File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/process.py", line 121, in start self._popen = self._Popen(self) File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/context.py", line 284, in _Popen return Popen(process_obj) File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/popen_spawn_posix.py", line 32, in __init__ super().__init__(process_obj) File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/popen_fork.py", line 19, in __init__ self._launch(process_obj) File "/Users/x/opt/anaconda3/lib/python3.9/multiprocessing/popen_spawn_posix.py", line 47, in _xxlaunch reduction.dump(process_obj, fp) File "/Users/xxx/opt/anaconda3/lib/python3.9/multiprocessing/reduction.py", line 60, in dump ForkingPickler(file, protocol).dump(obj) TypeError: cannot pickle 'weakref' object
问题原因
将类方法self.t_test作为multiprocessing.Process的target时,整个类实例会被序列化(pickle)传递给子进程。类实例中包含的multiprocessing.Manager()对象内部存在无法被pickle序列化的弱引用结构;同时df.columns返回的pd.Index对象也可能包含弱引用,双重因素导致序列化失败。
解决方案
方案1:使用进程池替代手动创建Process,分离任务逻辑
把t_test逻辑改为独立函数,通过进程池批量执行任务,用返回值收集结果,避免传递整个类实例:
import multiprocessing from scipy import stats import numpy as np import pandas as pd def t_test(args): data, classes, k = args index_samples = data[:, k] # 用numpy布尔索引替代列表推导,提升效率 mask = classes == "Virginia" rs1 = index_samples[mask] rs2 = index_samples[~mask] return stats.ttest_ind(rs1, rs2, equal_var=False).pvalue def main(): df = pd.read_excel("/Users/xxx/Documents/Project/src/flattened.xlsx") flattened = df.values.T # 将pd.Index转为普通列表,避免序列化问题 y = df.columns.tolist() # 用进程池自动管理子进程 with multiprocessing.Pool() as pool: tasks = [(flattened, y, k) for k in range(10)] pval = pool.map(t_test, tasks) print(pval) if __name__ == "__main__": main()
方案2:保留类结构,解耦序列化依赖
通过静态方法+队列传递结果,避免序列化整个类实例,只传递必要的可序列化数据:
import multiprocessing from scipy import stats import numpy as np import pandas as pd class T_TestFeature: def __init__(self, data, classes): self.data = data self.classes = classes self.pval = [] def preform(self): with multiprocessing.Manager() as manager: result_queue = manager.Queue() processes = [] for i in range(10): p = multiprocessing.Process( target=self._t_test_task, args=(self.data, self.classes, i, result_queue) ) processes.append(p) p.start() for p in processes: p.join() # 从队列收集结果 while not result_queue.empty(): self.pval.append(result_queue.get()) @staticmethod def _t_test_task(data, classes, k, result_queue): index_samples = data[:, k] mask = classes == "Virginia" rs1 = index_samples[mask] rs2 = index_samples[~mask] result_queue.put(stats.ttest_ind(rs1, rs2, equal_var=False).pvalue) def main(): df = pd.read_excel("/Users/xxx/Documents/Project/src/flattened.xlsx") flattened = df.values.T y = df.columns.tolist() result = T_TestFeature(flattened, y) result.preform() print(result.pval) if __name__ == "__main__": main()
关键优化点
- 将
pd.Index转为普通列表(.tolist()),规避索引对象的序列化问题 - 用numpy布尔索引替代列表推导,提升数据筛选效率
- 通过进程池或队列传递结果,避免共享类实例中的Manager对象,减少序列化依赖
内容的提问来源于stack exchange,提问作者Grid Varavithya Vitvara
相关产品推荐
相关产品推荐

