You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C#中如何用FFmpeg等编码器逐帧编码视频并支持音频写入

在C#中基于FFmpeg实现逐帧低内存视频编码(含音频支持)

用FFmpeg.AutoGen这个C#绑定库可以直接调用FFmpeg原生API,实现逐帧编码(处理完单帧立即释放内存),同时支持音频采样写入。下面是具体实现步骤和代码示例:

1. 环境准备

通过NuGet安装FFmpeg.AutoGen包,同时确保系统中安装对应版本的FFmpeg运行库(或者将FFmpeg的dll文件放到项目输出目录)。

2. 核心实现逻辑

  • 初始化FFmpeg全局环境,创建输出文件的格式上下文和编码上下文
  • 对每个RGB帧:将byte[]格式的24位RGB数据转换为FFmpeg要求的像素格式(如YUV420P),生成单帧AVFrame,编码后写入文件,立即释放该帧的内存
  • 音频部分:将音频采样(如PCM格式)逐块转换为FFmpeg的音频帧,编码写入,同样逐块释放内存
  • 编码完成后写入文件尾,清理所有FFmpeg上下文资源

3. 代码示例

3.1 初始化FFmpeg和输出上下文

using FFmpeg.AutoGen;
using System;
using System.IO;

public class FrameByFrameEncoder
{
    private AVFormatContext* _formatContext;
    private AVCodecContext* _videoCodecContext;
    private AVCodecContext* _audioCodecContext;
    private SwsContext* _swsContext;
    private int _videoStreamIndex;
    private int _audioStreamIndex;
    private long _videoPts;
    private long _audioPts;

    // 视频参数:根据实际需求调整
    private const int Width = 1920;
    private const int Height = 1080;
    private const int FrameRate = 30;
    private const int BitRate = 8000000;

    // 音频参数:根据实际需求调整
    private const int AudioSampleRate = 44100;
    private const int AudioChannels = 2;
    private const int AudioBitRate = 128000;

    public void Initialize(string outputPath)
    {
        // 初始化FFmpeg
        ffmpeg.av_register_all();
        ffmpeg.avformat_network_init();

        // 创建输出格式上下文
        ffmpeg.avformat_alloc_output_context2(&_formatContext, null, null, outputPath);
        if (_formatContext == null)
            throw new InvalidOperationException("无法创建输出格式上下文");

        // 添加视频流
        AddVideoStream();
        // 添加音频流
        AddAudioStream();

        // 打开输出文件并写入头信息
        ffmpeg.avio_open(&_formatContext->pb, outputPath, ffmpeg.AVIO_FLAG_WRITE);
        ffmpeg.avformat_write_header(_formatContext, null);
    }

    private void AddVideoStream()
    {
        var codec = ffmpeg.avcodec_find_encoder(_formatContext->oformat->video_codec);
        if (codec == null)
            throw new InvalidOperationException("找不到视频编码器");

        _videoCodecContext = ffmpeg.avcodec_alloc_context3(codec);
        _videoCodecContext->codec_id = codec->id;
        _videoCodecContext->codec_type = AVMediaType.AVMEDIA_TYPE_VIDEO;
        _videoCodecContext->width = Width;
        _videoCodecContext->height = Height;
        _videoCodecContext->pix_fmt = AVPixelFormat.AV_PIX_FMT_YUV420P; // 大部分编码器支持的格式
        _videoCodecContext->bit_rate = BitRate;
        _videoCodecContext->time_base = new AVRational { num = 1, den = FrameRate };
        _videoCodecContext->framerate = new AVRational { num = FrameRate, den = 1 };

        // 如果是MP4等格式,需要设置全局头
        if ((_formatContext->oformat->flags & ffmpeg.AVFMT_GLOBALHEADER) != 0)
            _videoCodecContext->flags |= ffmpeg.AV_CODEC_FLAG_GLOBAL_HEADER;

        // 打开编码器
        if (ffmpeg.avcodec_open2(_videoCodecContext, codec, null) < 0)
            throw new InvalidOperationException("无法打开视频编码器");

        // 添加流到格式上下文
        var videoStream = ffmpeg.avformat_new_stream(_formatContext, codec);
        _videoStreamIndex = videoStream->index;
        ffmpeg.avcodec_parameters_from_context(videoStream->codecpar, _videoCodecContext);
    }

    private void AddAudioStream()
    {
        var codec = ffmpeg.avcodec_find_encoder(_formatContext->oformat->audio_codec);
        if (codec == null)
            throw new InvalidOperationException("找不到音频编码器");

        _audioCodecContext = ffmpeg.avcodec_alloc_context3(codec);
        _audioCodecContext->codec_id = codec->id;
        _audioCodecContext->codec_type = AVMediaType.AVMEDIA_TYPE_AUDIO;
        _audioCodecContext->sample_rate = AudioSampleRate;
        _audioCodecContext->channels = AudioChannels;
        _audioCodecContext->channel_layout = ffmpeg.av_get_default_channel_layout(AudioChannels);
        _audioCodecContext->sample_fmt = codec->sample_fmts[0]; // 使用编码器支持的第一个采样格式
        _audioCodecContext->bit_rate = AudioBitRate;
        _audioCodecContext->time_base = new AVRational { num = 1, den = AudioSampleRate };

        if ((_formatContext->oformat->flags & ffmpeg.AVFMT_GLOBALHEADER) != 0)
            _audioCodecContext->flags |= ffmpeg.AV_CODEC_FLAG_GLOBAL_HEADER;

        if (ffmpeg.avcodec_open2(_audioCodecContext, codec, null) < 0)
            throw new InvalidOperationException("无法打开音频编码器");

        var audioStream = ffmpeg.avformat_new_stream(_formatContext, codec);
        _audioStreamIndex = audioStream->index;
        ffmpeg.avcodec_parameters_from_context(audioStream->codecpar, _audioCodecContext);
    }

3.2 逐帧编码RGB数据

public void EncodeVideoFrame(byte[] rgbFrame)
    {
        // 初始化像素转换上下文(第一次调用时创建)
        if (_swsContext == null)
        {
            _swsContext = ffmpeg.sws_getContext(
                Width, Height, AVPixelFormat.AV_PIX_FMT_RGB24,
                Width, Height, _videoCodecContext->pix_fmt,
                ffmpeg.SWS_BILINEAR, null, null, null);
        }

        // 创建AVFrame并填充RGB数据
        var frame = ffmpeg.av_frame_alloc();
        frame->width = Width;
        frame->height = Height;
        frame->format = (int)AVPixelFormat.AV_PIX_FMT_RGB24;
        ffmpeg.av_frame_get_buffer(frame, 0);
        ffmpeg.av_frame_make_writable(frame);

        // 将byte[]复制到AVFrame的缓冲区
        var bufferSize = ffmpeg.av_image_get_buffer_size(AVPixelFormat.AV_PIX_FMT_RGB24, Width, Height, 1);
        System.Runtime.InteropServices.Marshal.Copy(rgbFrame, 0, frame->data[0], bufferSize);

        // 转换像素格式为编码器要求的格式
        var outputFrame = ffmpeg.av_frame_alloc();
        outputFrame->width = Width;
        outputFrame->height = Height;
        outputFrame->format = (int)_videoCodecContext->pix_fmt;
        ffmpeg.av_frame_get_buffer(outputFrame, 0);
        ffmpeg.av_frame_make_writable(outputFrame);

        ffmpeg.sws_scale(_swsContext, frame->data, frame->linesize, 0, Height, outputFrame->data, outputFrame->linesize);

        // 设置时间戳
        outputFrame->pts = _videoPts++;
        ffmpeg.av_packet_unref(new AVPacket());
        var pkt = new AVPacket();
        ffmpeg.av_init_packet(&pkt);

        // 编码帧
        var sendResult = ffmpeg.avcodec_send_frame(_videoCodecContext, outputFrame);
        if (sendResult >= 0)
        {
            while (ffmpeg.avcodec_receive_packet(_videoCodecContext, &pkt) >= 0)
            {
                // 调整时间戳以匹配流的时间基
                ffmpeg.av_packet_rescale_ts(&pkt, _videoCodecContext->time_base, _formatContext->streams[_videoStreamIndex]->time_base);
                pkt.stream_index = _videoStreamIndex;
                // 写入文件
                ffmpeg.av_interleaved_write_frame(_formatContext, &pkt);
                ffmpeg.av_packet_unref(&pkt);
            }
        }

        // 释放当前帧的内存,避免累积占用
        ffmpeg.av_frame_free(&frame);
        ffmpeg.av_frame_free(&outputFrame);
    }

3.3 写入音频采样

public void EncodeAudioSamples(byte[] pcmSamples)
    {
        // 创建音频帧
        var frame = ffmpeg.av_frame_alloc();
        frame->sample_rate = AudioSampleRate;
        frame->channels = AudioChannels;
        frame->channel_layout = _audioCodecContext->channel_layout;
        frame->format = (int)_audioCodecContext->sample_fmt;

        // 计算采样数:PCM是16位的话,每个采样2字节,双通道则每个样本4字节
        var sampleSize = ffmpeg.av_get_bytes_per_sample((AVSampleFormat)_audioCodecContext->sample_fmt);
        var sampleCount = pcmSamples.Length / (sampleSize * AudioChannels);
        frame->nb_samples = sampleCount;

        ffmpeg.av_frame_get_buffer(frame, 0);
        ffmpeg.av_frame_make_writable(frame);

        // 复制PCM数据到帧缓冲区
        System.Runtime.InteropServices.Marshal.Copy(pcmSamples, 0, frame->data[0], pcmSamples.Length);

        // 设置时间戳
        frame->pts = _audioPts;
        _audioPts += sampleCount;

        ffmpeg.av_packet_unref(new AVPacket());
        var pkt = new AVPacket();
        ffmpeg.av_init_packet(&pkt);

        // 编码音频帧
        var sendResult = ffmpeg.avcodec_send_frame(_audioCodecContext, frame);
        if (sendResult >= 0)
        {
            while (ffmpeg.avcodec_receive_packet(_audioCodecContext, &pkt) >= 0)
            {
                ffmpeg.av_packet_rescale_ts(&pkt, _audioCodecContext->time_base, _formatContext->streams[_audioStreamIndex]->time_base);
                pkt.stream_index = _audioStreamIndex;
                ffmpeg.av_interleaved_write_frame(_formatContext, &pkt);
                ffmpeg.av_packet_unref(&pkt);
            }
        }

        // 释放音频帧内存
        ffmpeg.av_frame_free(&frame);
    }

3.4 收尾工作

public void FinalizeEncoding()
    {
        // 刷新编码器,确保所有剩余帧都写入
        FlushEncoder(_videoCodecContext, _videoStreamIndex);
        FlushEncoder(_audioCodecContext, _audioStreamIndex);

        // 写入文件尾
        ffmpeg.av_write_trailer(_formatContext);

        // 关闭编码器和文件
        ffmpeg.avcodec_close(_videoCodecContext);
        ffmpeg.avcodec_close(_audioCodecContext);
        ffmpeg.avio_close(_formatContext->pb);
        ffmpeg.avformat_free_context(_formatContext);
        if (_swsContext != null)
            ffmpeg.sws_freeContext(_swsContext);
    }

    private void FlushEncoder(AVCodecContext* codecContext, int streamIndex)
    {
        var pkt = new AVPacket();
        ffmpeg.av_init_packet(&pkt);
        while (ffmpeg.avcodec_send_frame(codecContext, null) >= 0)
        {
            while (ffmpeg.avcodec_receive_packet(codecContext, &pkt) >= 0)
            {
                ffmpeg.av_packet_rescale_ts(&pkt, codecContext->time_base, _formatContext->streams[streamIndex]->time_base);
                pkt.stream_index = streamIndex;
                ffmpeg.av_interleaved_write_frame(_formatContext, &pkt);
                ffmpeg.av_packet_unref(&pkt);
            }
        }
    }
}

4. 使用示例

public static void Main()
{
    var encoder = new FrameByFrameEncoder();
    encoder.Initialize("output.mp4");

    // 模拟逐帧读取/生成RGB帧
    for (int i = 0; i < 300; i++) // 10秒30fps视频
    {
        byte[] rgbFrame = GenerateTestRgbFrame(); // 替换为你的实际RGB帧数据
        encoder.EncodeVideoFrame(rgbFrame);
    }

    // 模拟写入音频采样
    byte[] audioSamples = GenerateTestAudioSamples(); // 替换为你的实际PCM音频数据
    encoder.EncodeAudioSamples(audioSamples);

    encoder.FinalizeEncoding();
}

// 测试用的RGB帧生成函数
private static byte[] GenerateTestRgbFrame()
{
    int width = 1920;
    int height = 1080;
    byte[] frame = new byte[width * height * 3];
    // 生成纯色帧,这里是红色
    for (int i = 0; i < frame.Length; i += 3)
    {
        frame[i] = 255;     // R
        frame[i + 1] = 0;   // G
        frame[i + 2] = 0;   // B
    }
    return frame;
}

// 测试用的音频采样生成函数
private static byte[] GenerateTestAudioSamples()
{
    int sampleCount = 44100 * 10; // 10秒音频
    byte[] samples = new byte[sampleCount * 2 * 2]; // 16位双声道
    // 生成正弦波测试音频
    double frequency = 440;
    double amplitude = 0.5;
    for (int i = 0; i < sampleCount; i++)
    {
        double t = (double)i / 44100;
        double sample = amplitude * Math.Sin(2 * Math.PI * frequency * t);
        short val = (short)(sample * short.MaxValue);
        // 写入左声道
        samples[i * 4] = (byte)(val & 0xFF);
        samples[i * 4 + 1] = (byte)((val >> 8) & 0xFF);
        // 写入右声道
        samples[i * 4 + 2] = (byte)(val & 0xFF);
        samples[i * 4 + 3] = (byte)((val >> 8) & 0xFF);
    }
    return samples;
}

关键注意事项

  • 内存管理:每帧处理完成后必须调用av_frame_free释放帧内存,避免内存泄漏;编码后的数据包要调用av_packet_unref释放。
  • 像素格式转换:大部分视频编码器只支持YUV格式(如YUV420P),所以必须将RGB24转换为对应格式,sws_scale是FFmpeg提供的高效转换函数。
  • 时间戳同步:视频帧和音频帧的pts必须正确设置,否则会出现音画不同步的问题,时间戳要根据流的时间基进行转换。
  • 编码器刷新:编码结束后必须调用FlushEncoder,确保编码器中剩余的帧全部写入文件。

内容的提问来源于stack exchange,提问作者Hippolippo

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.20 06:51:17