如何用FFmpeg实现RTSP直播动态字幕(无文件)及修复流异常
问题与解决方案:FFmpeg RTSP桌面流动态字幕实现
问题描述
作为大型项目的一部分,需通过FFmpeg推送RTSP桌面直播流,并根据场景动态切换字幕。已实现基础直播功能,但希望避免使用文本/临时文件,尝试通过生成带字幕的图片叠加到流中,当前代码仅在程序终止时才推送几秒流数据。
当前问题代码:
import subprocess import threading import string import random import time import io from PIL import Image, ImageDraw, ImageFont RTSP_URL = "..." ffmpeg = None def generate_subtitle(): width = 640 height = 100 font_size = 32 while True: if ffmpeg: try: text = ''.join(random.choice(string.ascii_uppercase + string.digits) for _ in range(10)) image = Image.new("RGBA", (width, height), (0, 0, 0, 128)) draw = ImageDraw.Draw(image) try: font = ImageFont.truetype("arial.ttf", font_size) except IOError: font = ImageFont.load_default() bbox = draw.textbbox((0, 0), text, font=font) text_width = bbox[2] - bbox[0] text_height = bbox[3] - bbox[1] x = (width - text_width) // 2 y = (height - text_height) // 2 draw.text((x, y), text, font=font, fill=(255, 255, 255, 255)) buffer = io.BytesIO() image.save(buffer, format="PNG") ffmpeg.stdin.write(buffer.getvalue()) ffmpeg.stdin.flush() time.sleep(5) except Exception as e: print("Erreur d'envoi d'image :", e) break else: time.sleep(1) def run_ffmpeg(): global ffmpeg ffmpeg = subprocess.Popen([ 'ffmpeg', # Input 0: capture desktop "-f", "gdigrab", "-offset_x", "0", "-offset_y", "0", "-video_size", "1920x1080", "-i", "desktop", # Input 1: PNG overlay from stdin "-f", "image2pipe", "-vcodec", "png", "-i", "-", # Filter to overlay Input 1 on Input 0 "-filter_complex", "[0:v][1:v]overlay=(main_w-overlay_w)/2:(main_h-overlay_h)-10", # Output settings "-vcodec", "libx264", "-preset", "ultrafast", "-tune", "zerolatency", "-g", "30", "-sc_threshold", "0", "-f", "rtsp", RTSP_URL ], stdin=subprocess.PIPE) threading.Thread(target=run_ffmpeg, daemon=True).start() threading.Thread(target=generate_subtitle, daemon=True).start() while True: time.sleep(1)
解决方案
一、修复图片叠加的流推送问题
当前代码的核心问题:
image2pipe输入未指定帧率,FFmpeg无法正确处理间隔发送的图片帧,引发缓冲;- 桌面采集输入未固定帧率,导致FFmpeg时钟同步混乱;
- subprocess的stdin默认有缓冲,图片数据无法实时传递给FFmpeg。
修改后的代码:
import subprocess import threading import string import random import time import io from PIL import Image, ImageDraw, ImageFont RTSP_URL = "..." ffmpeg = None def generate_subtitle(): width = 640 height = 100 font_size = 32 while True: if ffmpeg and ffmpeg.poll() is None: try: text = ''.join(random.choice(string.ascii_uppercase + string.digits) for _ in range(10)) image = Image.new("RGBA", (width, height), (0, 0, 0, 128)) draw = ImageDraw.Draw(image) try: font = ImageFont.truetype("arial.ttf", font_size) except IOError: font = ImageFont.load_default() bbox = draw.textbbox((0, 0), text, font=font) text_width = bbox[2] - bbox[0] text_height = bbox[3] - bbox[1] x = (width - text_width) // 2 y = (height - text_height) // 2 draw.text((x, y), text, font=font, fill=(255, 255, 255, 255)) buffer = io.BytesIO() image.save(buffer, format="PNG") # 写入数据并强制无缓冲发送 ffmpeg.stdin.write(buffer.getvalue()) ffmpeg.stdin.flush() time.sleep(5) except Exception as e: print("图片发送错误:", e) break else: time.sleep(1) def run_ffmpeg(): global ffmpeg # 给桌面采集固定帧率,给图片输入设置匹配的帧率(每5秒1帧) ffmpeg = subprocess.Popen([ 'ffmpeg', # 桌面采集输入:固定帧率30 "-f", "gdigrab", "-framerate", "30", "-offset_x", "0", "-offset_y", "0", "-video_size", "1920x1080", "-i", "desktop", # 图片管道输入:设置帧率为1/5(每5秒1帧),确保FFmpeg正确处理间隔输入 "-f", "image2pipe", "-r", "1/5", "-vcodec", "png", "-i", "-", # 滤镜:添加shortest=0确保以桌面流为主,同时指定输出格式兼容RTSP "-filter_complex", "[0:v][1:v]overlay=(main_w-overlay_w)/2:(main_h-overlay_h)-10,format=yuv420p", # 输出设置:保持低延迟配置 "-vcodec", "libx264", "-preset", "ultrafast", "-tune", "zerolatency", "-g", "30", "-sc_threshold", "0", "-f", "rtsp", RTSP_URL ], stdin=subprocess.PIPE, bufsize=0) # 设置bufsize=0实现无缓冲IO threading.Thread(target=run_ffmpeg, daemon=True).start() threading.Thread(target=generate_subtitle, daemon=True).start() while True: # 检查FFmpeg进程状态,异常退出则终止主程序 if ffmpeg and ffmpeg.poll() is not None: print("FFmpeg进程异常退出") break time.sleep(1)
二、无需临时文件的动态字幕替代方案
直接使用FFmpeg的drawtext滤镜结合管道传递动态文本,无需生成图片,更高效:
实现思路
- 使用
drawtext滤镜的textfile参数读取管道中的文本内容; - Python通过管道向FFmpeg实时发送字幕文本,每次更新时写入新文本即可。
代码实现(Windows环境适配命名管道)
import subprocess import threading import string import random import time import win32pipe import win32file import pywintypes RTSP_URL = "..." ffmpeg = None def generate_subtitle(pipe_handle): while True: if ffmpeg and ffmpeg.poll() is None: try: text = ''.join(random.choice(string.ascii_uppercase + string.digits) for _ in range(10)) # 写入文本到命名管道,注意编码和换行 win32file.WriteFile(pipe_handle, (text + "\n").encode('utf-8')) time.sleep(5) except Exception as e: print("字幕发送错误:", e) break else: time.sleep(1) def run_ffmpeg(): global ffmpeg # 创建Windows命名管道 pipe_name = r'\\.\pipe\ffmpeg_subtitle_pipe' pipe_handle = win32pipe.CreateNamedPipe( pipe_name, win32pipe.PIPE_ACCESS_OUTBOUND, win32pipe.PIPE_TYPE_MESSAGE | win32pipe.PIPE_WAIT, 1, 65536, 65536, 0, None ) # 等待FFmpeg连接管道 win32pipe.ConnectNamedPipe(pipe_handle, None) # FFmpeg命令:使用drawtext滤镜读取命名管道的文本 ffmpeg = subprocess.Popen([ 'ffmpeg', "-f", "gdigrab", "-framerate", "30", "-offset_x", "0", "-offset_y", "0", "-video_size", "1920x1080", "-i", "desktop", "-vf", f"drawtext=textfile='{pipe_name}':x=(w-text_w)/2:y=h-text_h-10:fontsize=32:fontcolor=white:box=1:boxcolor=black@0.5:reload=1", "-vcodec", "libx264", "-preset", "ultrafast", "-tune", "zerolatency", "-g", "30", "-sc_threshold", "0", "-f", "rtsp", RTSP_URL ]) # 启动字幕生成线程 threading.Thread(target=generate_subtitle, args=(pipe_handle,), daemon=True).start() # 启动FFmpeg线程 threading.Thread(target=run_ffmpeg, daemon=True).start() while True: if ffmpeg and ffmpeg.poll() is not None: print("FFmpeg进程异常退出") break time.sleep(1)
说明
reload=1参数让drawtext滤镜实时读取管道的最新文本内容;- Linux/macOS环境可使用匿名管道替代命名管道,实现更简洁;
- 该方案无需生成图片,减少CPU开销,字幕更新更高效。
内容的提问来源于stack exchange,提问作者Rorschy
相关产品推荐
相关产品推荐

