YOLOv8集成Google TTS时遭遇AttributeError问题求助
问题描述
我做的项目是用YOLOv8检测目标标签和坐标,把标签转成字符串后用gTTS生成语音,但获取预测标签时一直报AttributeError,刚接触这个框架,求帮忙。
原代码
import cv2 from gtts import gTTS import os from ultralytics import YOLO def convert_labels_to_text(labels): text = ", ".join(labels) return text class YOLOWithLabels(YOLO): def __call__(self, frame): results = super().__call__(frame) labels = results.pred[0].get_field("labels").tolist() annotated_frame = results.render() return annotated_frame, labels cap = cv2.VideoCapture(0) model = YOLOWithLabels('yolov8n.pt') while cap.isOpened(): success, frame = cap.read() if success: annotated_frame, labels = model(frame) message = convert_labels_to_text(labels) tts_engine = gTTS(text=message) # Initialize gTTS with the message tts_engine.save("output.mp3") os.system("output.mp3") cv2.putText(annotated_frame, message, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 2) cv2.imshow("YOLOv8 Inference", annotated_frame) if cv2.waitKey(1) & 0xFF == ord("q"): break else: break cap.release() cv2.destroyAllWindows()
错误信息
File "C:\Users\alien\Desktop\YOLOv8 project files\gtts service\testservice.py", line 13, in __call__ labels = results.pred[0].get_field("labels").tolist() ^^^^^^^^^^^^ AttributeError: 'list' object has no attribute 'pred'
打印results的输出
orig_shape: (480, 640) path: 'image0.jpg' probs: None save_dir: None speed: {'preprocess': 3.1604766845703125, 'inference': 307.905912399292, 'postprocess': 2.8924942016601562}] 0: 480x640 1 person, 272.4ms Speed: 3.0ms preprocess, 272.4ms inference, 4.0ms postprocess per image at shape (1, 3, 640, 640) [ultralytics.yolo.engine.results.Results object with attributes: boxes: ultralytics.yolo.engine.results.Boxes object keypoints: None keys: ['boxes'] masks: None names: {0: 'person', 1: 'bicycle', 2: 'car', 3: 'motorcycle', 4: 'airplane', 5: 'bus', 6: 'train', 7: 'truck', 8: 'boat', 9: 'traffic light', 10: 'fire hydrant', 11: 'stop sign', 12: 'parking meter', 13: 'bench', 14: 'bird', 15: 'cat', 16: 'dog', 17: 'horse', 18: 'sheep', 19: 'cow', 20: 'elephant', 21: 'bear', 22: 'zebra', 23: 'giraffe', 24: 'backpack', 25: 'umbrella', 26: 'handbag', 27: 'tie', 28: 'suitcase', 29: 'frisbee', 30: 'skis', 31: 'snowboard', 32: 'sports ball', 33: 'kite', 34: 'baseball bat', 35: 'baseball glove', 36: 'skateboard', 37: 'surfboard', 38: 'tennis racket', 39: 'bottle', 40: 'wine glass', 41: 'cup', 42: 'fork', 43: 'knife', 44: 'spoon', 45: 'bowl', 46: 'banana', 47: 'apple', 48: 'sandwich', 49: 'orange', 50: 'broccoli', 51: 'carrot', 52: 'hot dog', 53: 'pizza', 54: 'donut', 55: 'cake', 56: 'chair', 57: 'couch', 58: 'potted plant', 59: 'bed', 60: 'dining table', 61: 'toilet', 62: 'tv', 63: 'laptop', 64: 'mouse', 65: 'remote', 66: 'keyboard', 67: 'cell phone', 68: 'microwave', 69: 'oven', 70: 'toaster', 71: 'sink', 72: 'refrigerator', 73: 'book', 74: 'clock', 75: 'vase', 76: 'scissors', 77: 'teddy bear', 78: 'hair drier', 79: 'toothbrush'} orig_img: array([[[168, 167, 166], [165, 165, 165], [165, 166, 167], ..., [183, 186, 178], [183, 186, 178], [184, 187, 179]], [[168, 167, 165], [166, 165, 165], [166, 167, 166], ..., [184, 187, 179], [183, 186, 178], [184, 187, 179]], [[168, 167, 164], [167, 167, 164], [167, 167, 165], ..., [184, 187, 178], [184, 187, 179], [183, 186, 178]], ..., [[196, 192, 185], [196, 192, 185], [196, 192, 185], ..., [ 25, 29, 38], [ 22, 25, 35], [ 20, 24, 34]], [[199, 195, 187], [197, 193, 186], [197, 193, 186], ..., [ 23, 26, 35], [ 22, 25, 35], [ 22, 25, 35]], [[199, 195, 187], [199, 195, 187], [199, 195, 187], ..., [ 20, 24, 33], [ 19, 23, 33], [ 19, 23, 33]]], dtype=uint8)
解决方案
问题出在YOLOv8的API使用上,你用的是旧版本YOLO的写法,YOLOv8的__call__方法返回的是Results对象的列表,而不是单个Results对象,而且获取标签的方式也变了。
修正后的代码如下:
import cv2 from gtts import gTTS import os from ultralytics import YOLO def convert_labels_to_text(labels): text = ", ".join(labels) return text class YOLOWithLabels(YOLO): def __call__(self, frame): results = super().__call__(frame) # 取列表中的第一个Results对象 result = results[0] # 获取类别ID,再通过names映射成标签名称 class_ids = result.boxes.cls.tolist() labels = [result.names[int(id)] for id in class_ids] # 渲染带标注的帧 annotated_frame = result.plot() return annotated_frame, labels cap = cv2.VideoCapture(0) model = YOLOWithLabels('yolov8n.pt') while cap.isOpened(): success, frame = cap.read() if success: annotated_frame, labels = model(frame) message = convert_labels_to_text(labels) # 只有检测到目标时才生成语音,避免空文本报错 if message: tts_engine = gTTS(text=message) tts_engine.save("output.mp3") os.system("output.mp3") cv2.putText(annotated_frame, message, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 2) cv2.imshow("YOLOv8 Inference", annotated_frame) if cv2.waitKey(1) & 0xFF == ord("q"): break else: break cap.release() cv2.destroyAllWindows()
关键修改点
- 处理results返回值:
super().__call__(frame)返回的是Results对象列表,需要取第一个元素result = results[0] - 正确获取标签:YOLOv8中,检测框的类别ID存在
result.boxes.cls里,通过result.names字典可以把ID转换成对应的标签名称 - 渲染标注帧:用
result.plot()替代旧的results.render(),这是YOLOv8的标准渲染方法 - 增加空判断:当没有检测到目标时,message为空,此时不执行gTTS操作,避免报错
内容的提问来源于stack exchange,提问作者asfriendlyascarbon
相关产品推荐
相关产品推荐

