如何修复YOLOv3处理MP4时出现的IndexError: invalid index to scalar variable?
YOLOv3处理MP4视频报错修复方案
问题描述
使用YOLOv3处理MP4视频时,持续触发以下错误:
invalid index to scalar variable
错误指向获取YOLO输出层名称的代码段,原代码如下:
import numpy as np import cv2 # initialize minimum probability to eliminate weak predictions p_min = 0.5 # threshold when applying non-maxia suppression thres = 0. # 'VideoCapture' object and reading video from a file video = cv2.VideoCapture('parking1.mp4') # Preparing variable for writer # that we will use to write processed frames writer = None # Preparing variables for spatial dimensions of the frames h, w = None, None # Create labels into list with open('coco.names') as f: labels = [line.strip() for line in f] # Initialize colours for representing every detected object colours = np.random.randint(0, 255, size=(len(labels), 3), dtype='uint8') # Loading trained YOLO v3 Objects Detector # with the help of 'dnn' library from OpenCV # Reads a network model stored in Darknet model files. network = cv2.dnn.readNetFromDarknet('yolov3.cfg', 'yolov3.weights') # Getting only output layer names that we need from YOLO(ERRORRRRRRRRRR HEREEE) ln = network.getLayerNames() ln = [ln[i[0] - 1] for i in network.getUnconnectedOutLayers()] # Defining loop for catching frames while True: ret, frame = video.read() if not ret: break # Getting dimensions of the frame for once as everytime dimensions will be same if w is None or h is None: # Slicing and get height, width of the image h, w = frame.shape[:2] # frame preprocessing for deep learning blob = cv2.dnn.blobFromImage(frame, 1 / 255.0, (416, 416), swapRB=True, crop=False)
错误截图显示代码执行到ln = [ln[i[0] - 1] for i in network.getUnconnectedOutLayers()]时抛出异常。
错误原因
不同OpenCV版本中,getUnconnectedOutLayers()的返回值类型不一致:
- 旧版本返回标量数组(如
array([200, 227, 254])) - 新版本返回二维数组(如
array([[200], [227], [254]]))
原代码中i[0]的写法仅适配新版本,若使用旧版OpenCV,会因标量无法索引而报错。
修复方案
统一处理两种返回值类型,修改获取输出层名称的代码:
ln = network.getLayerNames() # 处理不同版本的返回值格式 out_layers = network.getUnconnectedOutLayers() # 判断返回值是标量还是数组 if isinstance(out_layers[0], list): ln = [ln[i[0] - 1] for i in out_layers] else: ln = [ln[i - 1] for i in out_layers]
完整优化代码
以下是可直接运行的完整YOLOv3视频处理代码,包含检测、绘制结果、保存输出视频的完整流程:
import numpy as np import cv2 # 参数配置 p_min = 0.5 # 最小置信度阈值 thres = 0.3 # NMS非极大值抑制阈值 input_video = 'parking1.mp4' output_video = 'parking_detected.mp4' # 加载视频 video = cv2.VideoCapture(input_video) if not video.isOpened(): print(f"无法打开视频文件: {input_video}") exit() # 初始化视频写入器 writer = None h, w = None, None # 加载类别标签 with open('coco.names', 'r') as f: labels = [line.strip() for line in f.readlines()] # 生成类别颜色 colours = np.random.randint(0, 255, size=(len(labels), 3), dtype='uint8') # 加载YOLOv3模型 network = cv2.dnn.readNetFromDarknet('yolov3.cfg', 'yolov3.weights') # 设置使用GPU加速(可选,需CUDA版本OpenCV) # network.setPreferableBackend(cv2.dnn.DNN_BACKEND_CUDA) # network.setPreferableTarget(cv2.dnn.DNN_TARGET_CUDA) # 获取输出层名称(修复版本) ln = network.getLayerNames() out_layers = network.getUnconnectedOutLayers() if isinstance(out_layers[0], list): ln = [ln[i[0] - 1] for i in out_layers] else: ln = [ln[i - 1] for i in out_layers] # 逐帧处理视频 while True: ret, frame = video.read() if not ret: break # 获取帧尺寸 if w is None or h is None: h, w = frame.shape[:2] # 帧预处理 blob = cv2.dnn.blobFromImage(frame, 1/255.0, (416, 416), swapRB=True, crop=False) network.setInput(blob) outputs = network.forward(ln) # 存储检测结果 boxes = [] confidences = [] class_ids = [] # 解析输出结果 for output in outputs: for detection in output: scores = detection[5:] class_id = np.argmax(scores) confidence = scores[class_id] if confidence > p_min: # 转换为原始帧坐标 box = detection[0:4] * np.array([w, h, w, h]) centerX, centerY, width, height = box.astype('int') x = int(centerX - (width / 2)) y = int(centerY - (height / 2)) boxes.append([x, y, int(width), int(height)]) confidences.append(float(confidence)) class_ids.append(class_id) # 非极大值抑制去除重复框 indices = cv2.dnn.NMSBoxes(boxes, confidences, p_min, thres) # 绘制检测结果 if len(indices) > 0: for i in indices.flatten(): x, y, w_box, h_box = boxes[i] color = colours[class_ids[i]].tolist() label = f"{labels[class_ids[i]]}: {confidences[i]:.2f}" # 绘制边框 cv2.rectangle(frame, (x, y), (x + w_box, y + h_box), color, 2) # 绘制标签背景 cv2.rectangle(frame, (x, y - 20), (x + len(label)*10, y), color, -1) # 绘制标签文本 cv2.putText(frame, label, (x, y - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0,0,0), 2) # 初始化视频写入器 if writer is None: fourcc = cv2.VideoWriter_fourcc(*'mp4v') writer = cv2.VideoWriter(output_video, fourcc, 30, (frame.shape[1], frame.shape[0]), True) # 写入处理后的帧 writer.write(frame) # 实时显示(可选,按q退出) cv2.imshow('YOLOv3 Video Detection', frame) if cv2.waitKey(1) & 0xFF == ord('q'): break # 释放资源 video.release() if writer is not None: writer.release() cv2.destroyAllWindows()
注意事项
- 确保
yolov3.cfg、yolov3.weights、coco.names文件路径正确,可从YOLO官方仓库下载 - 若运行缓慢,可开启GPU加速(需安装CUDA版本的OpenCV)
- 调整
p_min和thres参数可平衡检测精度与速度
内容的提问来源于stack exchange,提问作者Maryam
相关产品推荐
相关产品推荐

