大数据场景下Python多进程崩溃:OSError: [Errno 24] 打开文件过多
问题描述
我用Python3.8通过多进程将约700GB的.mp4数据集转换为图片(每秒提取一帧),程序会读取数据集路径,创建dataset_image文件夹并保留原目录结构。在8核、128GB内存的环境运行约8.5小时(CPU时间超60小时)后,程序崩溃。尝试添加/移除try-catch块没有改善,图片保存和目录结构都正常。
运行代码
import os import sys import cv2 from pathlib import Path from multiprocessing import Process base_path = 'path_to_dataset' images_dir = 'path_to_dataset_images' image_fps = 1 def task(path, name): filename, file_extension = os.path.splitext(os.path.relpath(os.path.join(path, name), base_path)) if file_extension != '.mp4': exit(0) video_path = os.path.join(path, name) image_path = os.path.join(images_dir, filename) extension = ".png" try: cap = cv2.VideoCapture(video_path) fps = round(cap.get(cv2.CAP_PROP_FPS)) hop = round(fps / image_fps) curr_frame = 0 while True: ret, frame = cap.read() if not ret: break if curr_frame % hop == 0: name = image_path + "_" + str(curr_frame) + extension Path(os.path.dirname(name)).mkdir(parents=True, exist_ok=True) if cv2.imwrite(name, frame): print(name + " saved") else: print(name + " not saved!", file=sys.stderr) curr_frame += 1 except: print(image_path + " failed due to error", file=sys.stderr) finally: cap.release() def convert_to_image(): # create all tasks processes = [] for path, subdirs, files in os.walk(base_path): processes = processes + [Process(target=task, args=(path, name,)) for name in files] # start all processes for process in processes: process.start() # wait for all processes to complete for process in processes: process.join() # report that all tasks are completed print('Done', flush=True) if __name__ == "__main__": convert_to_image()
错误信息
Traceback (most recent call last): File "convert_to_images.py", line 72, in <module> convert_to_image() File "convert_to_images.py", line 59, in convert_to_image process.start() File "/home/user/.conda/envs/tf2.9/lib/python3.8/multiprocessing/process.py", line 121, in start self._popen = self._Popen(self) File "/home/user/.conda/envs/tf2.9/lib/python3.8/multiprocessing/context.py", line 224, in _Popen return _default_context.get_context().Process._Popen(process_obj) File "/home/user/.conda/envs/tf2.9/lib/python3.8/multiprocessing/context.py", line 277, in _Popen return Popen(process_obj) File "/home/user/.conda/envs/tf2.9/lib/python3.8/multiprocessing/popen_fork.py", line 19, in __init__ self._launch(process_obj) File "/home/user/.conda/envs/tf2.9/lib/python3.8/multiprocessing/popen_fork.py", line 69, in _launch child_r, parent_w = os.pipe() OSError: [Errno 24] Too many open files
问题分析与修复方案
问题根源
当前代码会为每个视频文件创建一个独立进程,当数据集文件数量极大时,会同时启动大量进程。每个进程创建时会占用管道文件描述符,再加上OpenCV读取视频、写入图片时打开的文件,最终超出系统允许的最大打开文件数限制,触发Errno 24错误。
修复方案
1. 用进程池限制并发数
不要一次性启动所有进程,改用multiprocessing.Pool设置与CPU核心数匹配的最大并发数(比如8核就设为8),避免同时创建过多进程占用资源。
2. 优化资源管理
确保视频流、文件等资源在异常情况下也能正确释放,避免资源泄漏。
修改后的代码示例
import os import sys import cv2 from pathlib import Path from multiprocessing import Pool base_path = 'path_to_dataset' images_dir = 'path_to_dataset_images' image_fps = 1 MAX_WORKERS = 8 # 与CPU核心数匹配 def task(args): path, name = args filename, file_extension = os.path.splitext(os.path.relpath(os.path.join(path, name), base_path)) if file_extension != '.mp4': return video_path = os.path.join(path, name) image_path = os.path.join(images_dir, filename) extension = ".png" cap = None try: cap = cv2.VideoCapture(video_path) if not cap.isOpened(): print(f"Failed to open video: {video_path}", file=sys.stderr) return fps = round(cap.get(cv2.CAP_PROP_FPS)) hop = round(fps / image_fps) curr_frame = 0 while True: ret, frame = cap.read() if not ret: break if curr_frame % hop == 0: img_name = f"{image_path}_{curr_frame}{extension}" img_dir = os.path.dirname(img_name) Path(img_dir).mkdir(parents=True, exist_ok=True) if cv2.imwrite(img_name, frame): print(f"{img_name} saved") else: print(f"{img_name} not saved!", file=sys.stderr) curr_frame += 1 except Exception as e: print(f"{image_path} failed due to error: {str(e)}", file=sys.stderr) finally: if cap is not None: cap.release() def convert_to_image(): # 生成任务列表 tasks = [] for path, subdirs, files in os.walk(base_path): for name in files: tasks.append((path, name)) # 使用进程池处理任务 with Pool(MAX_WORKERS) as pool: pool.map(task, tasks) print('Done', flush=True) if __name__ == "__main__": convert_to_image()
修改说明
- 改用
Pool控制最大并发数,避免一次性启动大量进程导致资源耗尽 - 增加视频流初始化检查,避免未打开视频时执行
release()报错 - 捕获具体异常并打印错误信息,方便调试
- 使用f-string格式化字符串,代码更简洁易读
内容的提问来源于stack exchange,提问作者Tomas Lapsansky
相关产品推荐
相关产品推荐

