如何用FFmpeg实现MP3/WAV批量播放并显示文件名,修复音频编码异常
修复FFmpeg生成视频时的单声道音频与音频中断问题
问题概述
使用Python脚本批量将指定文件夹中的MP3/WAV文件合成带文件名显示的视频时,出现两个核心问题:
- 合成后的视频音频仅左声道输出
- 最终视频的音频存在中断现象
问题原因
- 单声道问题:原脚本的音频重编码命令未强制输出双声道格式,若源音频本身为单声道,FFmpeg会默认保留单声道编码,导致最终视频仅单声道播放。
- 音频中断问题:原脚本使用源音频时长生成视频帧的时长参数,但重编码后的音频可能因编码精度存在微小时长偏差,导致音视频时长不匹配;同时原合成命令的
-shortest参数会在音视频时长未完全对齐时提前截断音频。
修改后的完整脚本
import os import subprocess def create_combined_video(audio_folder, output_path): # 获取指定文件夹中的音频文件列表 audio_files = [] for filename in os.listdir(audio_folder): if filename.endswith(".mp3") or filename.endswith(".wav"): audio_files.append(os.path.join(audio_folder, filename)) # 按文件名排序 audio_files.sort() # 创建临时帧文件夹 temp_frames_folder = "temp_frames" os.makedirs(temp_frames_folder, exist_ok=True) # 生成带文件名的图片帧 for index, audio_file in enumerate(audio_files): name = os.path.splitext(os.path.basename(audio_file))[0] image_path = os.path.join(temp_frames_folder, f"{index+1:06d}.jpg") # FFmpeg生成带文字的单帧图片 ffmpeg_cmd = f'ffmpeg -y -f lavfi -i color=c=white:s=720x480:d=1 -vf "drawtext=text=\'{name}\':fontcolor=black:fontsize=36:x=(w-text_w)/2:y=(h-text_h)/2" -vframes 1 "{image_path}"' subprocess.run(ffmpeg_cmd, shell=True, check=True) # 重新编码音频为AAC双声道 reencoded_audio_folder = "reencoded_audio" os.makedirs(reencoded_audio_folder, exist_ok=True) reencoded_audio_list = [] for index, audio_file in enumerate(audio_files): reencoded_audio_file = os.path.join(reencoded_audio_folder, f"{index:03d}.m4a") # 强制双声道输出,确保音频为立体声 ffmpeg_cmd = f'ffmpeg -y -i "{audio_file}" -c:a aac -ac 2 "{reencoded_audio_file}"' subprocess.run(ffmpeg_cmd, shell=True, check=True) reencoded_audio_list.append(reencoded_audio_file) # 生成图片帧列表文件(使用重编码后的音频时长) image_names_path = "image_names.txt" with open(image_names_path, "w") as file: for index, audio_file in enumerate(reencoded_audio_list): image_path = os.path.join(temp_frames_folder, f"{index+1:06d}.jpg") duration = get_audio_duration(audio_file) file.write(f"file '{image_path}'\nduration {duration}\n") # 最后一行额外添加一次最后一张图片,避免concat截断问题 file.write(f"file '{os.path.join(temp_frames_folder, f'{len(reencoded_audio_list):06d}.jpg')}'\n") # 生成重编码音频列表文件 reencoded_audio_names_path = "reencoded_audio_names.txt" with open(reencoded_audio_names_path, "w") as file: for audio_file in reencoded_audio_list: file.write(f"file '{audio_file}'\n") # 合成最终视频 ffmpeg_cmd = f'ffmpeg -y -f concat -safe 0 -i "{image_names_path}" -f concat -safe 0 -i "{reencoded_audio_names_path}" -c:v libx264 -pix_fmt yuv420p -vf "scale=720:480:force_original_aspect_ratio=increase,crop=720:480" -c:a aac -async 1 "{output_path}"' subprocess.run(ffmpeg_cmd, shell=True, check=True) # 清理临时文件 os.remove(image_names_path) os.remove(reencoded_audio_names_path) for image_file in os.listdir(temp_frames_folder): os.remove(os.path.join(temp_frames_folder, image_file)) os.rmdir(temp_frames_folder) for audio_file in os.listdir(reencoded_audio_folder): os.remove(os.path.join(reencoded_audio_folder, audio_file)) os.rmdir(reencoded_audio_folder) def get_audio_duration(audio_file): # 使用FFprobe获取音频时长 ffprobe_cmd = f'ffprobe -v error -show_entries format=duration -of default=noprint_wrappers=1:nokey=1 "{audio_file}"' result = subprocess.run(ffprobe_cmd, shell=True, capture_output=True, text=True, check=True) return float(result.stdout.strip()) # 使用示例 audio_folder = "C:/Users/Admin/Desktop/Sounds" output_path = "C:/Users/Admin/Desktop/output.mp4" create_combined_video(audio_folder, output_path)
关键修改点说明
- 强制双声道编码:在音频重编码命令中添加
-ac 2参数,强制输出立体声格式,彻底解决单声道问题。 - 对齐音视频时长:改用重编码后的音频时长生成视频帧的duration参数,避免源音频与重编码音频的时长偏差;在图片帧列表末尾额外添加一次最后一张图片,解决FFmpeg concat协议对最后一帧的截断问题。
- 优化合成同步逻辑:将
-shortest替换为-async 1,强制音频与视频同步,避免音频被提前截断;为subprocess.run()添加check=True,确保命令执行出错时及时抛出异常,便于调试。 - 清理冗余代码:移除未使用的
audio_names.txt生成逻辑,简化脚本结构。
内容的提问来源于stack exchange,提问作者iiDk
相关产品推荐
相关产品推荐

