如何用OpenCV Python实现带音频的视频录制、编辑输出与播放?
问题描述
已实现摄像头捕获视频保存到指定目录的功能,但使用cv2.VideoWriter保存的编辑后视频无音频,且OpenCV默认播放视频时也没有声音。现有代码如下:
import cv2 import numpy as np video = cv2.VideoCapture(0); if video.isOpened(): # get vcap property width = int(video.get(cv2.CAP_PROP_FRAME_WIDTH)) # float `width` height = int(video.get(cv2.CAP_PROP_FRAME_HEIGHT)) # float `width` output = cv2.VideoWriter('edited.mp4', cv2.VideoWriter_fourcc('m','p','4','v'), 30, (width, height)) while True: grabbed, frame = video.read(); if grabbed == True: hh, ww = frame.shape[:2] w, h = (6, 6) result = cv2.resize(frame, (w, h), interpolation=cv2.INTER_AREA) result_enl = cv2.resize(result, (ww, hh), interpolation=cv2.INTER_AREA) colors_ = []; for i, row in enumerate(result): my_array = np.array(result[i]); max_values = my_array.max(0); r, g, b = (int(max_values[0]),int(max_values[1]),int(max_values[2])); cx, cy = int(width/2), int(height/2)-300; # creating each rectangle of color size = (15, 100); middle_size = int(size[1]); top_left = (30, cy + (i*size[1])); bottom_right = (size[0], cy + (size[1] * (i+1))); cv2.rectangle(frame, top_left, bottom_right, (r, g, b), -1); cv2.putText(frame, str(i), (30, cy + ((i*size[1]) + 50)), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (36,255,12), 2) video_output = cv2.resize(frame,(width,height)); output.write(video_output); cv2.imshow("frame", frame); else: break; key = cv2.waitKey(1) & 0xFF # if the `q` key is pressed, break from the loop if key == ord("q"): break output.release(); video.release(); cv2.destroyAllWindows()
核心原因
OpenCV本身不支持音频处理,不管是cv2.VideoCapture还是cv2.VideoWriter都只负责视频流的读写,不会处理音频数据,所以需要借助其他库来实现音频的捕获、播放和合并。
一、实现带音频的视频播放
方案1:使用MoviePy播放
MoviePy基于FFmpeg,可直接读取并播放带音频的视频:
from moviepy.editor import VideoFileClip # 播放本地带音频的视频 clip = VideoFileClip("edited_with_audio.mp4") clip.preview() # 自带音视频同步播放窗口
方案2:使用Pygame播放
适合实时音视频场景,需配合音频捕获逻辑:
import pygame import cv2 # 初始化Pygame pygame.init() screen = pygame.display.set_mode((width, height)) clock = pygame.time.Clock() # 读取本地带音频视频(实时场景需额外捕获音频) video = cv2.VideoCapture("edited_with_audio.mp4") # 若音频分离,可单独加载播放 # audio = pygame.mixer.Sound("audio.wav") # audio.play() running = True while running: for event in pygame.event.get(): if event.type == pygame.QUIT: running = False ret, frame = video.read() if not ret: break # 转换OpenCV的BGR格式为Pygame的RGB格式 frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) frame_surface = pygame.surfarray.make_surface(frame.swapaxes(0,1)) screen.blit(frame_surface, (0,0)) pygame.display.flip() clock.tick(30) video.release() pygame.quit()
二、输出带音频的编辑后视频
方案1:使用MoviePy(推荐)
先分别捕获处理视频、录制音频,再合并两者:
import cv2 import numpy as np import pyaudio import wave from moviepy.editor import VideoFileClip, AudioFileClip # 配置参数 FORMAT = pyaudio.paInt16 CHANNELS = 2 RATE = 44100 CHUNK = 1024 TEMP_AUDIO = "temp_audio.wav" TEMP_VIDEO = "temp_video.mp4" FINAL_VIDEO = "edited_with_audio.mp4" # 初始化音频捕获 audio = pyaudio.PyAudio() stream = audio.open(format=FORMAT, channels=CHANNELS, rate=RATE, input=True, frames_per_buffer=CHUNK) audio_frames = [] # 处理视频并保存临时文件 video = cv2.VideoCapture(0) if video.isOpened(): width = int(video.get(cv2.CAP_PROP_FRAME_WIDTH)) height = int(video.get(cv2.CAP_PROP_FRAME_HEIGHT)) output = cv2.VideoWriter(TEMP_VIDEO, cv2.VideoWriter_fourcc('m','p','4','v'), 30, (width, height)) print("录制中,按q停止...") running = True while running: grabbed, frame = video.read() if grabbed: # 原视频处理逻辑 hh, ww = frame.shape[:2] w, h = (6, 6) result = cv2.resize(frame, (w, h), interpolation=cv2.INTER_AREA) for i, row in enumerate(result): my_array = np.array(result[i]) max_values = my_array.max(0) r, g, b = (int(max_values[0]), int(max_values[1]), int(max_values[2])) cy = int(height/2) - 300 size = (15, 100) top_left = (30, cy + (i*size[1])) bottom_right = (size[0]+30, cy + (size[1]*(i+1))) cv2.rectangle(frame, top_left, bottom_right, (r,g,b), -1) cv2.putText(frame, str(i), (30, cy + (i*size[1]+50)), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (36,255,12), 2) output.write(frame) cv2.imshow("frame", frame) # 捕获音频帧 audio_data = stream.read(CHUNK) audio_frames.append(audio_data) else: break key = cv2.waitKey(1) & 0xFF if key == ord("q"): running = False # 释放资源 output.release() video.release() cv2.destroyAllWindows() stream.stop_stream() stream.close() audio.terminate() # 保存音频文件 wf = wave.open(TEMP_AUDIO, 'wb') wf.setnchannels(CHANNELS) wf.setsampwidth(audio.get_sample_size(FORMAT)) wf.setframerate(RATE) wf.writeframes(b''.join(audio_frames)) wf.close() # 合并音视频 video_clip = VideoFileClip(TEMP_VIDEO) audio_clip = AudioFileClip(TEMP_AUDIO) final_clip = video_clip.set_audio(audio_clip) final_clip.write_videofile(FINAL_VIDEO, codec='libx264', audio_codec='aac') # 清理临时文件(可选) import os os.remove(TEMP_VIDEO) os.remove(TEMP_AUDIO)
方案2:使用FFmpeg命令行合并
先保存无音频视频和WAV音频,再执行命令合并:
ffmpeg -i temp_video.mp4 -i temp_audio.wav -c:v copy -c:a aac edited_with_audio.mp4
需确保系统已安装FFmpeg并添加到环境变量。
内容的提问来源于stack exchange,提问作者plus
相关产品推荐
相关产品推荐

