C#中如何用FFmpeg等编码器逐帧编码视频并支持音频写入
在C#中基于FFmpeg实现逐帧低内存视频编码(含音频支持)
用FFmpeg.AutoGen这个C#绑定库可以直接调用FFmpeg原生API,实现逐帧编码(处理完单帧立即释放内存),同时支持音频采样写入。下面是具体实现步骤和代码示例:
1. 环境准备
通过NuGet安装FFmpeg.AutoGen包,同时确保系统中安装对应版本的FFmpeg运行库(或者将FFmpeg的dll文件放到项目输出目录)。
2. 核心实现逻辑
- 初始化FFmpeg全局环境,创建输出文件的格式上下文和编码上下文
- 对每个RGB帧:将
byte[]格式的24位RGB数据转换为FFmpeg要求的像素格式(如YUV420P),生成单帧AVFrame,编码后写入文件,立即释放该帧的内存 - 音频部分:将音频采样(如PCM格式)逐块转换为FFmpeg的音频帧,编码写入,同样逐块释放内存
- 编码完成后写入文件尾,清理所有FFmpeg上下文资源
3. 代码示例
3.1 初始化FFmpeg和输出上下文
using FFmpeg.AutoGen; using System; using System.IO; public class FrameByFrameEncoder { private AVFormatContext* _formatContext; private AVCodecContext* _videoCodecContext; private AVCodecContext* _audioCodecContext; private SwsContext* _swsContext; private int _videoStreamIndex; private int _audioStreamIndex; private long _videoPts; private long _audioPts; // 视频参数:根据实际需求调整 private const int Width = 1920; private const int Height = 1080; private const int FrameRate = 30; private const int BitRate = 8000000; // 音频参数:根据实际需求调整 private const int AudioSampleRate = 44100; private const int AudioChannels = 2; private const int AudioBitRate = 128000; public void Initialize(string outputPath) { // 初始化FFmpeg ffmpeg.av_register_all(); ffmpeg.avformat_network_init(); // 创建输出格式上下文 ffmpeg.avformat_alloc_output_context2(&_formatContext, null, null, outputPath); if (_formatContext == null) throw new InvalidOperationException("无法创建输出格式上下文"); // 添加视频流 AddVideoStream(); // 添加音频流 AddAudioStream(); // 打开输出文件并写入头信息 ffmpeg.avio_open(&_formatContext->pb, outputPath, ffmpeg.AVIO_FLAG_WRITE); ffmpeg.avformat_write_header(_formatContext, null); } private void AddVideoStream() { var codec = ffmpeg.avcodec_find_encoder(_formatContext->oformat->video_codec); if (codec == null) throw new InvalidOperationException("找不到视频编码器"); _videoCodecContext = ffmpeg.avcodec_alloc_context3(codec); _videoCodecContext->codec_id = codec->id; _videoCodecContext->codec_type = AVMediaType.AVMEDIA_TYPE_VIDEO; _videoCodecContext->width = Width; _videoCodecContext->height = Height; _videoCodecContext->pix_fmt = AVPixelFormat.AV_PIX_FMT_YUV420P; // 大部分编码器支持的格式 _videoCodecContext->bit_rate = BitRate; _videoCodecContext->time_base = new AVRational { num = 1, den = FrameRate }; _videoCodecContext->framerate = new AVRational { num = FrameRate, den = 1 }; // 如果是MP4等格式,需要设置全局头 if ((_formatContext->oformat->flags & ffmpeg.AVFMT_GLOBALHEADER) != 0) _videoCodecContext->flags |= ffmpeg.AV_CODEC_FLAG_GLOBAL_HEADER; // 打开编码器 if (ffmpeg.avcodec_open2(_videoCodecContext, codec, null) < 0) throw new InvalidOperationException("无法打开视频编码器"); // 添加流到格式上下文 var videoStream = ffmpeg.avformat_new_stream(_formatContext, codec); _videoStreamIndex = videoStream->index; ffmpeg.avcodec_parameters_from_context(videoStream->codecpar, _videoCodecContext); } private void AddAudioStream() { var codec = ffmpeg.avcodec_find_encoder(_formatContext->oformat->audio_codec); if (codec == null) throw new InvalidOperationException("找不到音频编码器"); _audioCodecContext = ffmpeg.avcodec_alloc_context3(codec); _audioCodecContext->codec_id = codec->id; _audioCodecContext->codec_type = AVMediaType.AVMEDIA_TYPE_AUDIO; _audioCodecContext->sample_rate = AudioSampleRate; _audioCodecContext->channels = AudioChannels; _audioCodecContext->channel_layout = ffmpeg.av_get_default_channel_layout(AudioChannels); _audioCodecContext->sample_fmt = codec->sample_fmts[0]; // 使用编码器支持的第一个采样格式 _audioCodecContext->bit_rate = AudioBitRate; _audioCodecContext->time_base = new AVRational { num = 1, den = AudioSampleRate }; if ((_formatContext->oformat->flags & ffmpeg.AVFMT_GLOBALHEADER) != 0) _audioCodecContext->flags |= ffmpeg.AV_CODEC_FLAG_GLOBAL_HEADER; if (ffmpeg.avcodec_open2(_audioCodecContext, codec, null) < 0) throw new InvalidOperationException("无法打开音频编码器"); var audioStream = ffmpeg.avformat_new_stream(_formatContext, codec); _audioStreamIndex = audioStream->index; ffmpeg.avcodec_parameters_from_context(audioStream->codecpar, _audioCodecContext); }
3.2 逐帧编码RGB数据
public void EncodeVideoFrame(byte[] rgbFrame) { // 初始化像素转换上下文(第一次调用时创建) if (_swsContext == null) { _swsContext = ffmpeg.sws_getContext( Width, Height, AVPixelFormat.AV_PIX_FMT_RGB24, Width, Height, _videoCodecContext->pix_fmt, ffmpeg.SWS_BILINEAR, null, null, null); } // 创建AVFrame并填充RGB数据 var frame = ffmpeg.av_frame_alloc(); frame->width = Width; frame->height = Height; frame->format = (int)AVPixelFormat.AV_PIX_FMT_RGB24; ffmpeg.av_frame_get_buffer(frame, 0); ffmpeg.av_frame_make_writable(frame); // 将byte[]复制到AVFrame的缓冲区 var bufferSize = ffmpeg.av_image_get_buffer_size(AVPixelFormat.AV_PIX_FMT_RGB24, Width, Height, 1); System.Runtime.InteropServices.Marshal.Copy(rgbFrame, 0, frame->data[0], bufferSize); // 转换像素格式为编码器要求的格式 var outputFrame = ffmpeg.av_frame_alloc(); outputFrame->width = Width; outputFrame->height = Height; outputFrame->format = (int)_videoCodecContext->pix_fmt; ffmpeg.av_frame_get_buffer(outputFrame, 0); ffmpeg.av_frame_make_writable(outputFrame); ffmpeg.sws_scale(_swsContext, frame->data, frame->linesize, 0, Height, outputFrame->data, outputFrame->linesize); // 设置时间戳 outputFrame->pts = _videoPts++; ffmpeg.av_packet_unref(new AVPacket()); var pkt = new AVPacket(); ffmpeg.av_init_packet(&pkt); // 编码帧 var sendResult = ffmpeg.avcodec_send_frame(_videoCodecContext, outputFrame); if (sendResult >= 0) { while (ffmpeg.avcodec_receive_packet(_videoCodecContext, &pkt) >= 0) { // 调整时间戳以匹配流的时间基 ffmpeg.av_packet_rescale_ts(&pkt, _videoCodecContext->time_base, _formatContext->streams[_videoStreamIndex]->time_base); pkt.stream_index = _videoStreamIndex; // 写入文件 ffmpeg.av_interleaved_write_frame(_formatContext, &pkt); ffmpeg.av_packet_unref(&pkt); } } // 释放当前帧的内存,避免累积占用 ffmpeg.av_frame_free(&frame); ffmpeg.av_frame_free(&outputFrame); }
3.3 写入音频采样
public void EncodeAudioSamples(byte[] pcmSamples) { // 创建音频帧 var frame = ffmpeg.av_frame_alloc(); frame->sample_rate = AudioSampleRate; frame->channels = AudioChannels; frame->channel_layout = _audioCodecContext->channel_layout; frame->format = (int)_audioCodecContext->sample_fmt; // 计算采样数:PCM是16位的话,每个采样2字节,双通道则每个样本4字节 var sampleSize = ffmpeg.av_get_bytes_per_sample((AVSampleFormat)_audioCodecContext->sample_fmt); var sampleCount = pcmSamples.Length / (sampleSize * AudioChannels); frame->nb_samples = sampleCount; ffmpeg.av_frame_get_buffer(frame, 0); ffmpeg.av_frame_make_writable(frame); // 复制PCM数据到帧缓冲区 System.Runtime.InteropServices.Marshal.Copy(pcmSamples, 0, frame->data[0], pcmSamples.Length); // 设置时间戳 frame->pts = _audioPts; _audioPts += sampleCount; ffmpeg.av_packet_unref(new AVPacket()); var pkt = new AVPacket(); ffmpeg.av_init_packet(&pkt); // 编码音频帧 var sendResult = ffmpeg.avcodec_send_frame(_audioCodecContext, frame); if (sendResult >= 0) { while (ffmpeg.avcodec_receive_packet(_audioCodecContext, &pkt) >= 0) { ffmpeg.av_packet_rescale_ts(&pkt, _audioCodecContext->time_base, _formatContext->streams[_audioStreamIndex]->time_base); pkt.stream_index = _audioStreamIndex; ffmpeg.av_interleaved_write_frame(_formatContext, &pkt); ffmpeg.av_packet_unref(&pkt); } } // 释放音频帧内存 ffmpeg.av_frame_free(&frame); }
3.4 收尾工作
public void FinalizeEncoding() { // 刷新编码器,确保所有剩余帧都写入 FlushEncoder(_videoCodecContext, _videoStreamIndex); FlushEncoder(_audioCodecContext, _audioStreamIndex); // 写入文件尾 ffmpeg.av_write_trailer(_formatContext); // 关闭编码器和文件 ffmpeg.avcodec_close(_videoCodecContext); ffmpeg.avcodec_close(_audioCodecContext); ffmpeg.avio_close(_formatContext->pb); ffmpeg.avformat_free_context(_formatContext); if (_swsContext != null) ffmpeg.sws_freeContext(_swsContext); } private void FlushEncoder(AVCodecContext* codecContext, int streamIndex) { var pkt = new AVPacket(); ffmpeg.av_init_packet(&pkt); while (ffmpeg.avcodec_send_frame(codecContext, null) >= 0) { while (ffmpeg.avcodec_receive_packet(codecContext, &pkt) >= 0) { ffmpeg.av_packet_rescale_ts(&pkt, codecContext->time_base, _formatContext->streams[streamIndex]->time_base); pkt.stream_index = streamIndex; ffmpeg.av_interleaved_write_frame(_formatContext, &pkt); ffmpeg.av_packet_unref(&pkt); } } } }
4. 使用示例
public static void Main() { var encoder = new FrameByFrameEncoder(); encoder.Initialize("output.mp4"); // 模拟逐帧读取/生成RGB帧 for (int i = 0; i < 300; i++) // 10秒30fps视频 { byte[] rgbFrame = GenerateTestRgbFrame(); // 替换为你的实际RGB帧数据 encoder.EncodeVideoFrame(rgbFrame); } // 模拟写入音频采样 byte[] audioSamples = GenerateTestAudioSamples(); // 替换为你的实际PCM音频数据 encoder.EncodeAudioSamples(audioSamples); encoder.FinalizeEncoding(); } // 测试用的RGB帧生成函数 private static byte[] GenerateTestRgbFrame() { int width = 1920; int height = 1080; byte[] frame = new byte[width * height * 3]; // 生成纯色帧,这里是红色 for (int i = 0; i < frame.Length; i += 3) { frame[i] = 255; // R frame[i + 1] = 0; // G frame[i + 2] = 0; // B } return frame; } // 测试用的音频采样生成函数 private static byte[] GenerateTestAudioSamples() { int sampleCount = 44100 * 10; // 10秒音频 byte[] samples = new byte[sampleCount * 2 * 2]; // 16位双声道 // 生成正弦波测试音频 double frequency = 440; double amplitude = 0.5; for (int i = 0; i < sampleCount; i++) { double t = (double)i / 44100; double sample = amplitude * Math.Sin(2 * Math.PI * frequency * t); short val = (short)(sample * short.MaxValue); // 写入左声道 samples[i * 4] = (byte)(val & 0xFF); samples[i * 4 + 1] = (byte)((val >> 8) & 0xFF); // 写入右声道 samples[i * 4 + 2] = (byte)(val & 0xFF); samples[i * 4 + 3] = (byte)((val >> 8) & 0xFF); } return samples; }
关键注意事项
- 内存管理:每帧处理完成后必须调用
av_frame_free释放帧内存,避免内存泄漏;编码后的数据包要调用av_packet_unref释放。 - 像素格式转换:大部分视频编码器只支持YUV格式(如YUV420P),所以必须将RGB24转换为对应格式,
sws_scale是FFmpeg提供的高效转换函数。 - 时间戳同步:视频帧和音频帧的
pts必须正确设置,否则会出现音画不同步的问题,时间戳要根据流的时间基进行转换。 - 编码器刷新:编码结束后必须调用
FlushEncoder,确保编码器中剩余的帧全部写入文件。
内容的提问来源于stack exchange,提问作者Hippolippo
相关产品推荐
相关产品推荐

