You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用OpenCV Python实现带音频的视频录制、编辑输出与播放?

问题描述

已实现摄像头捕获视频保存到指定目录的功能,但使用cv2.VideoWriter保存的编辑后视频无音频,且OpenCV默认播放视频时也没有声音。现有代码如下:

import cv2
import numpy as np

video = cv2.VideoCapture(0);

if video.isOpened(): 
    # get vcap property 
    width  = int(video.get(cv2.CAP_PROP_FRAME_WIDTH))  # float `width`
    height  = int(video.get(cv2.CAP_PROP_FRAME_HEIGHT))  # float `width`

output = cv2.VideoWriter('edited.mp4', cv2.VideoWriter_fourcc('m','p','4','v'), 30, (width, height))

while True:

        grabbed, frame = video.read();

        if grabbed == True:
                hh, ww = frame.shape[:2]
                w, h = (6, 6)
                result = cv2.resize(frame, (w, h), interpolation=cv2.INTER_AREA)
                result_enl = cv2.resize(result, (ww, hh), interpolation=cv2.INTER_AREA)
           
                colors_ = [];
                
                for i, row in enumerate(result):
                        my_array = np.array(result[i]);
                        max_values = my_array.max(0);
                        r, g, b = (int(max_values[0]),int(max_values[1]),int(max_values[2]));
                        cx, cy = int(width/2), int(height/2)-300;
                        
                        # creating each rectangle of color
                        size = (15, 100);
                        middle_size = int(size[1]);
                        top_left = (30, cy + (i*size[1]));
                        bottom_right = (size[0], cy + (size[1] * (i+1)));
                        cv2.rectangle(frame, top_left, bottom_right, (r, g, b), -1);
                        cv2.putText(frame, str(i), (30, cy + ((i*size[1]) + 50)), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (36,255,12), 2)
                
                video_output = cv2.resize(frame,(width,height));
                output.write(video_output); 
                cv2.imshow("frame", frame);
        else: 
                break;
        key = cv2.waitKey(1) & 0xFF
        
        # if the `q` key is pressed, break from the loop
        if key == ord("q"):
                break

output.release();
video.release();
cv2.destroyAllWindows()

核心原因

OpenCV本身不支持音频处理,不管是cv2.VideoCapture还是cv2.VideoWriter都只负责视频流的读写,不会处理音频数据,所以需要借助其他库来实现音频的捕获、播放和合并。


一、实现带音频的视频播放

方案1:使用MoviePy播放

MoviePy基于FFmpeg,可直接读取并播放带音频的视频:

from moviepy.editor import VideoFileClip

# 播放本地带音频的视频
clip = VideoFileClip("edited_with_audio.mp4")
clip.preview()  # 自带音视频同步播放窗口

方案2:使用Pygame播放

适合实时音视频场景,需配合音频捕获逻辑:

import pygame
import cv2

# 初始化Pygame
pygame.init()
screen = pygame.display.set_mode((width, height))
clock = pygame.time.Clock()

# 读取本地带音频视频(实时场景需额外捕获音频)
video = cv2.VideoCapture("edited_with_audio.mp4")
# 若音频分离,可单独加载播放
# audio = pygame.mixer.Sound("audio.wav")
# audio.play()

running = True
while running:
    for event in pygame.event.get():
        if event.type == pygame.QUIT:
            running = False
    
    ret, frame = video.read()
    if not ret:
        break
    
    # 转换OpenCV的BGR格式为Pygame的RGB格式
    frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
    frame_surface = pygame.surfarray.make_surface(frame.swapaxes(0,1))
    screen.blit(frame_surface, (0,0))
    
    pygame.display.flip()
    clock.tick(30)

video.release()
pygame.quit()

二、输出带音频的编辑后视频

方案1:使用MoviePy(推荐)

先分别捕获处理视频、录制音频,再合并两者:

import cv2
import numpy as np
import pyaudio
import wave
from moviepy.editor import VideoFileClip, AudioFileClip

# 配置参数
FORMAT = pyaudio.paInt16
CHANNELS = 2
RATE = 44100
CHUNK = 1024
TEMP_AUDIO = "temp_audio.wav"
TEMP_VIDEO = "temp_video.mp4"
FINAL_VIDEO = "edited_with_audio.mp4"

# 初始化音频捕获
audio = pyaudio.PyAudio()
stream = audio.open(format=FORMAT, channels=CHANNELS,
                    rate=RATE, input=True, frames_per_buffer=CHUNK)
audio_frames = []

# 处理视频并保存临时文件
video = cv2.VideoCapture(0)
if video.isOpened():
    width = int(video.get(cv2.CAP_PROP_FRAME_WIDTH))
    height = int(video.get(cv2.CAP_PROP_FRAME_HEIGHT))

output = cv2.VideoWriter(TEMP_VIDEO, cv2.VideoWriter_fourcc('m','p','4','v'), 30, (width, height))

print("录制中,按q停止...")
running = True
while running:
    grabbed, frame = video.read()
    if grabbed:
        # 原视频处理逻辑
        hh, ww = frame.shape[:2]
        w, h = (6, 6)
        result = cv2.resize(frame, (w, h), interpolation=cv2.INTER_AREA)
        
        for i, row in enumerate(result):
            my_array = np.array(result[i])
            max_values = my_array.max(0)
            r, g, b = (int(max_values[0]), int(max_values[1]), int(max_values[2]))
            cy = int(height/2) - 300
            size = (15, 100)
            top_left = (30, cy + (i*size[1]))
            bottom_right = (size[0]+30, cy + (size[1]*(i+1)))
            cv2.rectangle(frame, top_left, bottom_right, (r,g,b), -1)
            cv2.putText(frame, str(i), (30, cy + (i*size[1]+50)), 
                        cv2.FONT_HERSHEY_SIMPLEX, 0.9, (36,255,12), 2)
        
        output.write(frame)
        cv2.imshow("frame", frame)
        
        # 捕获音频帧
        audio_data = stream.read(CHUNK)
        audio_frames.append(audio_data)
    else:
        break
    
    key = cv2.waitKey(1) & 0xFF
    if key == ord("q"):
        running = False

# 释放资源
output.release()
video.release()
cv2.destroyAllWindows()
stream.stop_stream()
stream.close()
audio.terminate()

# 保存音频文件
wf = wave.open(TEMP_AUDIO, 'wb')
wf.setnchannels(CHANNELS)
wf.setsampwidth(audio.get_sample_size(FORMAT))
wf.setframerate(RATE)
wf.writeframes(b''.join(audio_frames))
wf.close()

# 合并音视频
video_clip = VideoFileClip(TEMP_VIDEO)
audio_clip = AudioFileClip(TEMP_AUDIO)
final_clip = video_clip.set_audio(audio_clip)
final_clip.write_videofile(FINAL_VIDEO, codec='libx264', audio_codec='aac')

# 清理临时文件(可选)
import os
os.remove(TEMP_VIDEO)
os.remove(TEMP_AUDIO)

方案2:使用FFmpeg命令行合并

先保存无音频视频和WAV音频,再执行命令合并:

ffmpeg -i temp_video.mp4 -i temp_audio.wav -c:v copy -c:a aac edited_with_audio.mp4

需确保系统已安装FFmpeg并添加到环境变量。


内容的提问来源于stack exchange,提问作者plus

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.18 02:41:05