You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用FFmpeg实现RTSP直播动态字幕(无文件)及修复流异常

问题与解决方案:FFmpeg RTSP桌面流动态字幕实现

问题描述

作为大型项目的一部分,需通过FFmpeg推送RTSP桌面直播流,并根据场景动态切换字幕。已实现基础直播功能,但希望避免使用文本/临时文件,尝试通过生成带字幕的图片叠加到流中,当前代码仅在程序终止时才推送几秒流数据。

当前问题代码:

import subprocess
import threading
import string
import random
import time
import io
from PIL import Image, ImageDraw, ImageFont

RTSP_URL = "..."
ffmpeg = None

def generate_subtitle():
    width = 640
    height = 100
    font_size = 32
    while True:
        if ffmpeg:
            try:
                text = ''.join(random.choice(string.ascii_uppercase + string.digits) for _ in range(10))

                image = Image.new("RGBA", (width, height), (0, 0, 0, 128))
                draw = ImageDraw.Draw(image)

                try:
                    font = ImageFont.truetype("arial.ttf", font_size)
                except IOError:
                    font = ImageFont.load_default()

                bbox = draw.textbbox((0, 0), text, font=font)
                text_width = bbox[2] - bbox[0]
                text_height = bbox[3] - bbox[1]

                x = (width - text_width) // 2
                y = (height - text_height) // 2

                draw.text((x, y), text, font=font, fill=(255, 255, 255, 255))
                buffer = io.BytesIO()
                image.save(buffer, format="PNG")
                ffmpeg.stdin.write(buffer.getvalue())
                ffmpeg.stdin.flush()
                time.sleep(5)
            except Exception as e:
                print("Erreur d'envoi d'image :", e)
                break
        else:
            time.sleep(1)

def run_ffmpeg():
    global ffmpeg
    ffmpeg = subprocess.Popen([
        'ffmpeg',

        # Input 0: capture desktop
        "-f", "gdigrab",
        "-offset_x", "0",
        "-offset_y", "0",
        "-video_size", "1920x1080",
        "-i", "desktop",

        # Input 1: PNG overlay from stdin
        "-f", "image2pipe",
        "-vcodec", "png",
        "-i", "-",

        # Filter to overlay Input 1 on Input 0
        "-filter_complex", "[0:v][1:v]overlay=(main_w-overlay_w)/2:(main_h-overlay_h)-10",

        # Output settings
        "-vcodec", "libx264",
        "-preset", "ultrafast",
        "-tune", "zerolatency",
        "-g", "30",
        "-sc_threshold", "0",
        "-f", "rtsp",
        RTSP_URL
    ], stdin=subprocess.PIPE)

threading.Thread(target=run_ffmpeg, daemon=True).start()

threading.Thread(target=generate_subtitle, daemon=True).start()

while True:
    time.sleep(1)

解决方案

一、修复图片叠加的流推送问题

当前代码的核心问题:

  1. image2pipe输入未指定帧率,FFmpeg无法正确处理间隔发送的图片帧,引发缓冲;
  2. 桌面采集输入未固定帧率,导致FFmpeg时钟同步混乱;
  3. subprocess的stdin默认有缓冲,图片数据无法实时传递给FFmpeg。

修改后的代码:

import subprocess
import threading
import string
import random
import time
import io
from PIL import Image, ImageDraw, ImageFont

RTSP_URL = "..."
ffmpeg = None

def generate_subtitle():
    width = 640
    height = 100
    font_size = 32
    while True:
        if ffmpeg and ffmpeg.poll() is None:
            try:
                text = ''.join(random.choice(string.ascii_uppercase + string.digits) for _ in range(10))

                image = Image.new("RGBA", (width, height), (0, 0, 0, 128))
                draw = ImageDraw.Draw(image)

                try:
                    font = ImageFont.truetype("arial.ttf", font_size)
                except IOError:
                    font = ImageFont.load_default()

                bbox = draw.textbbox((0, 0), text, font=font)
                text_width = bbox[2] - bbox[0]
                text_height = bbox[3] - bbox[1]

                x = (width - text_width) // 2
                y = (height - text_height) // 2

                draw.text((x, y), text, font=font, fill=(255, 255, 255, 255))
                buffer = io.BytesIO()
                image.save(buffer, format="PNG")
                # 写入数据并强制无缓冲发送
                ffmpeg.stdin.write(buffer.getvalue())
                ffmpeg.stdin.flush()
                time.sleep(5)
            except Exception as e:
                print("图片发送错误:", e)
                break
        else:
            time.sleep(1)

def run_ffmpeg():
    global ffmpeg
    # 给桌面采集固定帧率,给图片输入设置匹配的帧率(每5秒1帧)
    ffmpeg = subprocess.Popen([
        'ffmpeg',
        # 桌面采集输入:固定帧率30
        "-f", "gdigrab",
        "-framerate", "30",
        "-offset_x", "0",
        "-offset_y", "0",
        "-video_size", "1920x1080",
        "-i", "desktop",
        # 图片管道输入:设置帧率为1/5(每5秒1帧),确保FFmpeg正确处理间隔输入
        "-f", "image2pipe",
        "-r", "1/5",
        "-vcodec", "png",
        "-i", "-",
        # 滤镜:添加shortest=0确保以桌面流为主,同时指定输出格式兼容RTSP
        "-filter_complex", "[0:v][1:v]overlay=(main_w-overlay_w)/2:(main_h-overlay_h)-10,format=yuv420p",
        # 输出设置:保持低延迟配置
        "-vcodec", "libx264",
        "-preset", "ultrafast",
        "-tune", "zerolatency",
        "-g", "30",
        "-sc_threshold", "0",
        "-f", "rtsp",
        RTSP_URL
    ], stdin=subprocess.PIPE, bufsize=0)  # 设置bufsize=0实现无缓冲IO

threading.Thread(target=run_ffmpeg, daemon=True).start()
threading.Thread(target=generate_subtitle, daemon=True).start()

while True:
    # 检查FFmpeg进程状态,异常退出则终止主程序
    if ffmpeg and ffmpeg.poll() is not None:
        print("FFmpeg进程异常退出")
        break
    time.sleep(1)

二、无需临时文件的动态字幕替代方案

直接使用FFmpeg的drawtext滤镜结合管道传递动态文本,无需生成图片,更高效:

实现思路

  1. 使用drawtext滤镜的textfile参数读取管道中的文本内容;
  2. Python通过管道向FFmpeg实时发送字幕文本,每次更新时写入新文本即可。

代码实现(Windows环境适配命名管道)

import subprocess
import threading
import string
import random
import time
import win32pipe
import win32file
import pywintypes

RTSP_URL = "..."
ffmpeg = None

def generate_subtitle(pipe_handle):
    while True:
        if ffmpeg and ffmpeg.poll() is None:
            try:
                text = ''.join(random.choice(string.ascii_uppercase + string.digits) for _ in range(10))
                # 写入文本到命名管道,注意编码和换行
                win32file.WriteFile(pipe_handle, (text + "\n").encode('utf-8'))
                time.sleep(5)
            except Exception as e:
                print("字幕发送错误:", e)
                break
        else:
            time.sleep(1)

def run_ffmpeg():
    global ffmpeg
    # 创建Windows命名管道
    pipe_name = r'\\.\pipe\ffmpeg_subtitle_pipe'
    pipe_handle = win32pipe.CreateNamedPipe(
        pipe_name,
        win32pipe.PIPE_ACCESS_OUTBOUND,
        win32pipe.PIPE_TYPE_MESSAGE | win32pipe.PIPE_WAIT,
        1, 65536, 65536,
        0,
        None
    )
    # 等待FFmpeg连接管道
    win32pipe.ConnectNamedPipe(pipe_handle, None)

    # FFmpeg命令:使用drawtext滤镜读取命名管道的文本
    ffmpeg = subprocess.Popen([
        'ffmpeg',
        "-f", "gdigrab",
        "-framerate", "30",
        "-offset_x", "0",
        "-offset_y", "0",
        "-video_size", "1920x1080",
        "-i", "desktop",
        "-vf", f"drawtext=textfile='{pipe_name}':x=(w-text_w)/2:y=h-text_h-10:fontsize=32:fontcolor=white:box=1:boxcolor=black@0.5:reload=1",
        "-vcodec", "libx264",
        "-preset", "ultrafast",
        "-tune", "zerolatency",
        "-g", "30",
        "-sc_threshold", "0",
        "-f", "rtsp",
        RTSP_URL
    ])

    # 启动字幕生成线程
    threading.Thread(target=generate_subtitle, args=(pipe_handle,), daemon=True).start()

# 启动FFmpeg线程
threading.Thread(target=run_ffmpeg, daemon=True).start()

while True:
    if ffmpeg and ffmpeg.poll() is not None:
        print("FFmpeg进程异常退出")
        break
    time.sleep(1)

说明

  • reload=1参数让drawtext滤镜实时读取管道的最新文本内容;
  • Linux/macOS环境可使用匿名管道替代命名管道,实现更简洁;
  • 该方案无需生成图片,减少CPU开销,字幕更新更高效。

内容的提问来源于stack exchange,提问作者Rorschy

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.13 02:43:10