You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于FFmpeg实现多路UDP网络流音视频合成输出的技术问询

Hey there! Let's break down how to translate that ffmpeg.exe command you're testing into a working program using the FFmpeg libraries. Your workflow—capturing multi-channel UDP inputs, demuxing, video overlay, audio mixing, encoding, and muxing to a file—is a common but powerful pipeline, so let's walk through each step with practical code pointers.

1. Initialize FFmpeg Core

First, kick things off by initializing the FFmpeg libraries. Since you're working with network streams (UDP), don't skip the network initialization:

#include <libavformat/avformat.h>
#include <libavfilter/avfilter.h>
#include <libavcodec/avcodec.h>

int main() {
    // Initialize core components
    av_register_all();
    avformat_network_init();
    avfilter_register_all();
    av_log_set_level(AV_LOG_VERBOSE); // Critical for debugging issues

    // Rest of your code here...
}

2. Open & Demultiplex UDP Input Streams

This corresponds to the -i input0 -i input1 part of your command. You'll need to create an AVFormatContext for each UDP input, open it, and extract stream metadata:

// Example: Open two UDP inputs
AVFormatContext *input_fmts[2] = {NULL, NULL};
const char *input_urls[2] = {"udp://your-first-stream-url", "udp://your-second-stream-url"};

for (int i = 0; i < 2; i++) {
    // Optional: Set UDP-specific options like timeout
    AVDictionary *opts = NULL;
    av_dict_set(&opts, "timeout", "5000000", 0); // 5-second timeout
    av_dict_set(&opts, "buffer_size", "1000000", 0); // 1MB buffer

    if (avformat_open_input(&input_fmts[i], input_urls[i], NULL, &opts) != 0) {
        av_log(NULL, AV_LOG_ERROR, "Failed to open input %d\n", i);
        return -1;
    }
    av_dict_free(&opts);

    if (avformat_find_stream_info(input_fmts[i], NULL) < 0) {
        av_log(NULL, AV_LOG_ERROR, "Failed to get stream info for input %d\n", i);
        return -1;
    }

    // Log stream details (matches ffmpeg.exe's verbose output)
    av_dump_format(input_fmts[i], i, input_urls[i], 0);
}

3. Build the Filter Graph (Video Overlay + Audio Mixing)

This is where your -filter_complex logic lives. Parsing a filter string (just like your command) is the easiest way to replicate your existing pipeline:

AVFilterGraph *filter_graph = avfilter_graph_alloc();
char filter_desc[1024];
// Replace with your exact filter complex string from the command line
snprintf(filter_desc, sizeof(filter_desc), 
         "[0:v][1:v]overlay=x=10:y=10[v]; [0:a][1:a]amix=inputs=2:duration=longest[a]");

// Create filter input/output mappings
AVFilterInOut *inputs = avfilter_inout_alloc();
AVFilterInOut *outputs = avfilter_inout_alloc();

// Map input streams to filter inputs
inputs->name = av_strdup("0:v");
inputs->pad_idx = 0;
inputs->next = avfilter_inout_alloc();
inputs->next->name = av_strdup("1:v");
inputs->next->pad_idx = 0;
inputs->next->next = avfilter_inout_alloc();
inputs->next->next->name = av_strdup("0:a");
inputs->next->next->pad_idx = 0;
inputs->next->next->next = avfilter_inout_alloc();
inputs->next->next->next->name = av_strdup("1:a");
inputs->next->next->next->pad_idx = 0;
inputs->next->next->next->next = NULL;

// Map filter outputs to encoder inputs
outputs->name = av_strdup("v");
outputs->pad_idx = 0;
outputs->next = avfilter_inout_alloc();
outputs->next->name = av_strdup("a");
outputs->next->pad_idx = 0;
outputs->next->next = NULL;

// Parse and configure the filter graph
if (avfilter_graph_parse_ptr(filter_graph, filter_desc, &inputs, &outputs, NULL) < 0) {
    av_log(NULL, AV_LOG_ERROR, "Failed to parse filter graph\n");
    return -1;
}
if (avfilter_graph_config(filter_graph, NULL) < 0) {
    av_log(NULL, AV_LOG_ERROR, "Failed to configure filter graph\n");
    return -1;
}

// Get references to filter input/output pads for later use
AVFilterContext *video_src0 = avfilter_graph_get_filter(filter_graph, "0:v");
AVFilterContext *audio_src0 = avfilter_graph_get_filter(filter_graph, "0:a");
AVFilterContext *video_out = avfilter_graph_get_filter(filter_graph, "v");
AVFilterContext *audio_out = avfilter_graph_get_filter(filter_graph, "a");

4. Initialize Output Encoders & Muxer

Next, set up the output file and encoders. This defines your output format (e.g., MP4, MKV) and codec parameters:

AVFormatContext *output_fmt = NULL;
const char *output_url = "final_output.mp4";

// Allocate output context based on file extension
if (avformat_alloc_output_context2(&output_fmt, NULL, NULL, output_url) < 0) {
    av_log(NULL, AV_LOG_ERROR, "Failed to create output context\n");
    return -1;
}

// Create video output stream
AVStream *video_stream = avformat_new_stream(output_fmt, NULL);
AVCodecParameters *video_par = video_stream->codecpar;
video_par->codec_id = output_fmt->oformat->video_codec;
video_par->codec_type = AVMEDIA_TYPE_VIDEO;
video_par->width = video_out->outputs[0]->w;
video_par->height = video_out->outputs[0]->h;
video_par->format = video_out->outputs[0]->format;
video_par->bit_rate = 2500000; // 2.5Mbps, adjust to your needs
video_stream->time_base = av_inv_q(video_out->outputs[0]->time_base);

// Create audio output stream
AVStream *audio_stream = avformat_new_stream(output_fmt, NULL);
AVCodecParameters *audio_par = audio_stream->codecpar;
audio_par->codec_id = output_fmt->oformat->audio_codec;
audio_par->codec_type = AVMEDIA_TYPE_AUDIO;
audio_par->sample_rate = audio_out->outputs[0]->sample_rate;
audio_par->channels = audio_out->outputs[0]->channels;
audio_par->format = audio_out->outputs[0]->format;
audio_stream->time_base = av_inv_q(audio_out->outputs[0]->time_base);

// Open output file for writing
if (!(output_fmt->oformat->flags & AVFMT_NOFILE)) {
    if (avio_open(&output_fmt->pb, output_url, AVIO_FLAG_WRITE) < 0) {
        av_log(NULL, AV_LOG_ERROR, "Failed to open output file\n");
        return -1;
    }
}

// Write file header
if (avformat_write_header(output_fmt, NULL) < 0) {
    av_log(NULL, AV_LOG_ERROR, "Failed to write output header\n");
    return -1;
}

5. Main Processing Loop: Read → Filter → Encode → Mux

This is the heart of your program: reading frames from inputs, processing them through filters, encoding, and writing to the output:

AVPacket pkt;
AVFrame *frame = av_frame_alloc();
AVFrame *filtered_frame = av_frame_alloc();

while (1) {
    // Read packets from both inputs (simplified sync logic)
    int ret = av_read_frame(input_fmts[0], &pkt);
    if (ret < 0) {
        ret = av_read_frame(input_fmts[1], &pkt);
        if (ret < 0) break; // Both inputs exhausted
    }

    AVFormatContext *input_ctx = (pkt.stream_index < input_fmts[0]->nb_streams) ? input_fmts[0] : input_fmts[1];
    AVStream *stream = input_ctx->streams[pkt.stream_index];

    // Decode packet to frame
    avcodec_send_packet(stream->codec, &pkt);
    ret = avcodec_receive_frame(stream->codec, frame);
    if (ret < 0) {
        av_packet_unref(&pkt);
        continue; // Skip corrupted frames
    }

    // Send frame to appropriate filter input
    if (stream->codecpar->codec_type == AVMEDIA_TYPE_VIDEO) {
        av_buffersrc_add_frame_flags(video_src0, frame, AV_BUFFERSRC_FLAG_KEEP_REF);
    } else if (stream->codecpar->codec_type == AVMEDIA_TYPE_AUDIO) {
        av_buffersrc_add_frame_flags(audio_src0, frame, AV_BUFFERSRC_FLAG_KEEP_REF);
    }

    // Process filtered video frames
    while (av_buffersink_get_frame(video_out, filtered_frame) >= 0) {
        avcodec_send_frame(video_stream->codec, filtered_frame);
        while (avcodec_receive_packet(video_stream->codec, &pkt) >= 0) {
            av_packet_rescale_ts(&pkt, video_stream->codec->time_base, video_stream->time_base);
            pkt.stream_index = video_stream->index;
            av_interleaved_write_frame(output_fmt, &pkt);
            av_packet_unref(&pkt);
        }
        av_frame_unref(filtered_frame);
    }

    // Process filtered audio frames (repeat similar logic to video)
    while (av_buffersink_get_frame(audio_out, filtered_frame) >= 0) {
        avcodec_send_frame(audio_stream->codec, filtered_frame);
        while (avcodec_receive_packet(audio_stream->codec, &pkt) >= 0) {
            av_packet_rescale_ts(&pkt, audio_stream->codec->time_base, audio_stream->time_base);
            pkt.stream_index = audio_stream->index;
            av_interleaved_write_frame(output_fmt, &pkt);
            av_packet_unref(&pkt);
        }
        av_frame_unref(filtered_frame);
    }

    av_frame_unref(frame);
    av_packet_unref(&pkt);
}

// Flush remaining frames from encoders
avcodec_send_frame(video_stream->codec, NULL);
avcodec_send_frame(audio_stream->codec, NULL);
// Add code to receive and write final packets...

6. Cleanup Resources

Don't forget to free all allocated resources to avoid memory leaks:

// Write output trailer
av_write_trailer(output_fmt);

// Close inputs
for (int i = 0; i < 2; i++) {
    avformat_close_input(&input_fmts[i]);
}

// Close output
if (!(output_fmt->oformat->flags & AVFMT_NOFILE)) {
    avio_closep(&output_fmt->pb);
}
avformat_free_context(output_fmt);

// Free filter graph and mappings
avfilter_graph_free(&filter_graph);
avfilter_inout_free(&inputs);
avfilter_inout_free(&outputs);

// Free frames
av_frame_free(&frame);
av_frame_free(&filtered_frame);

avformat_network_deinit();
return 0;

Key Debugging Tip

Since you're already testing with ffmpeg.exe, run your command with -v verbose to see FFmpeg's internal logs. In your program, av_log_set_level(AV_LOG_VERBOSE) will give you the same detailed output—this is invaluable for tracking down filter graph issues, codec compatibility problems, or stream sync bugs.

内容的提问来源于stack exchange,提问作者Hristo Ivanov

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.25 08:36:29