Unity VR光流计算存储代码提速及CSV格式转换技术问询
Unity VR光流计算性能优化与数据存储方案
场景说明
在搭配VR头显(强制上限90FPS)的Unity项目中,运行以下光流计算代码时,无代码项目稳定90FPS,添加后需通过WaitForSeconds(0.2f)才能维持80FPS以上帧率。目标是实现每帧计算并保存光流图,或至少将延迟降至~0.01秒,当前已使用AsyncGPUReadback和WriteAsync。
核心问题:如何进一步加速代码?
附加问题:能否将光流图以连续行形式写入单个CSV文件,而非单独PNG?该方式是否更慢?
using System.Collections; using UnityEngine; using System.IO; using UnityEngine.Rendering; namespace OpticalFlowAlternative { public class OpticalFlow : MonoBehaviour { protected enum Pass { Flow = 0, DownSample = 1, BlurH = 2, BlurV = 3, Visualize = 4 }; public RenderTexture Flow { get { return resultBuffer; } } [SerializeField] protected Material flowMaterial; protected RenderTexture prevFrame, flowBuffer, resultBuffer, renderTexture, rt; public string customOutputFolderPath = ""; private string filepathforflow; private int imageCount = 0; int targetTextureWidth, targetTextureHeight; private EyeTrackingV2 eyeTracking; protected void Start () { eyeTracking = GameObject.Find("XR Rig").GetComponent<EyeTrackingV2>(); targetTextureWidth = Screen.width / 16; targetTextureHeight = Screen.height / 16; flowMaterial.SetFloat("_Ratio", 1f * Screen.height / Screen.width); renderTexture = new RenderTexture(targetTextureWidth, targetTextureHeight, 0); rt = new RenderTexture(Screen.width, Screen.height, 0); StartCoroutine("StartCapture"); } protected void LateUpdate() { eyeTracking.flowCount = imageCount; } protected void OnDestroy () { if(prevFrame != null) { prevFrame.Release(); prevFrame = null; flowBuffer.Release(); flowBuffer = null; rt.Release(); rt = null; renderTexture.Release(); renderTexture = null; } } IEnumerator StartCapture() { while (true) { yield return new WaitForSeconds(0.2f); ScreenCapture.CaptureScreenshotIntoRenderTexture(rt); //compensating for image flip Graphics.Blit(rt, renderTexture, new Vector2(1, -1), new Vector2(0, 1)); if (prevFrame == null) { Setup(targetTextureWidth, targetTextureHeight); Graphics.Blit(renderTexture, prevFrame); } flowMaterial.SetTexture("_PrevTex", prevFrame); //calculating motion flow frame here Graphics.Blit(renderTexture, flowBuffer, flowMaterial, (int)Pass.Flow); Graphics.Blit(renderTexture, prevFrame); AsyncGPUReadback.Request(flowBuffer, 0, TextureFormat.ARGB32, OnCompleteReadback); } } void OnCompleteReadback(AsyncGPUReadbackRequest request) { if (request.hasError) return; var tex = new Texture2D(targetTextureWidth, targetTextureHeight, TextureFormat.ARGB32, false); tex.LoadRawTextureData(request.GetData<uint>()); tex.Apply(); WriteTextureAsync(tex); } async void WriteTextureAsync(Texture2D tex) { imageCount++; filepathforflow = customOutputFolderPath + imageCount + ".png"; var stream = new FileStream(filepathforflow, FileMode.OpenOrCreate); var bytes = tex.EncodeToPNG(); await stream.WriteAsync(bytes, 0, bytes.Length); } protected void Setup(int width, int height) { prevFrame = new RenderTexture(width, height, 0); prevFrame.format = RenderTextureFormat.ARGBFloat; prevFrame.wrapMode = TextureWrapMode.Repeat; prevFrame.Create(); flowBuffer = new RenderTexture(width, height, 0); flowBuffer.format = RenderTextureFormat.ARGBFloat; flowBuffer.wrapMode = TextureWrapMode.Repeat; flowBuffer.Create(); } } }
核心问题:代码加速方案
1. 替换高开销截图逻辑
ScreenCapture.CaptureScreenshotIntoRenderTexture会捕获全屏幕纹理再降采样,开销极大。建议直接从相机渲染到目标尺寸的RenderTexture:
- 在
Start中设置主相机的targetTexture = renderTexture,避免全屏幕渲染+降采样的双重开销。 - 若处理VR双眼画面,可针对左右眼相机分别设置渲染目标,或调用VR SDK提供的纹理获取接口。
2. 复用资源,减少GC与运行时创建
- 提前初始化所有
RenderTexture:将Setup调用移至Start方法,避免协程中动态创建纹理的开销。 - 复用
Texture2D对象:在Start中创建全局Texture2D,OnCompleteReadback中仅调用LoadRawTextureData和Apply,避免每帧创建新对象导致的GC。
3. 优化GPU读回与格式转换
- 匹配纹理格式:
flowBuffer使用RenderTextureFormat.ARGBFloat,将AsyncGPUReadback的请求格式改为TextureFormat.RGBAFloat,直接读取浮点数据,减少CPU端格式转换开销:AsyncGPUReadback.Request(flowBuffer, 0, TextureFormat.RGBAFloat, OnCompleteReadback); - 省略不必要的
Apply:若仅用于编码PNG,Texture2D.Apply()可跳过,因为LoadRawTextureData已直接写入像素数据。
4. 优化PNG编码与文件IO
- 使用
File.WriteAllBytesAsync替代手动创建FileStream,自动管理资源并减少IO开销:async void WriteTextureAsync(Texture2D tex) { imageCount++; filepathforflow = Path.Combine(customOutputFolderPath, $"{imageCount}.png"); var bytes = tex.EncodeToPNG(); await File.WriteAllBytesAsync(filepathforflow, bytes); } - 若Unity版本≥2020.1,使用
Texture2D.EncodeToPNGAsync异步编码,避免主线程阻塞:async void WriteTextureAsync(Texture2D tex) { imageCount++; filepathforflow = Path.Combine(customOutputFolderPath, $"{imageCount}.png"); var bytes = await tex.EncodeToPNGAsync(); await File.WriteAllBytesAsync(filepathforflow, bytes); }
5. 协程帧率控制优化
移除WaitForSeconds(0.2f),替换为yield return WaitForEndOfFrame(),确保每帧处理一次。若仍有帧率压力,可设置极小间隔(如yield return new WaitForSeconds(0.01f)),但优先优化前面的性能瓶颈。
6. Shader端优化
- 简化光流计算Shader:将
float精度改为half,减少GPU计算负载;移除未使用的Pass(如DownSample、Blur等)。 - 关闭RenderTexture的Mipmap与MSAA:创建
prevFrame和flowBuffer时,添加prevFrame.autoGenerateMips = false;并设置antiAliasing = 1。
附加问题:CSV存储方案与性能分析
可行性
完全可以将光流数据写入单个CSV文件。光流图的每个像素包含XY方向的运动向量(通常存储在RG通道),可将每帧的所有像素数据按行记录,格式示例:
FrameIndex,X,Y,FlowX,FlowY 1,0,0,0.12,-0.05 1,1,0,0.10,-0.03 ... 2,0,0,0.08,-0.06
性能对比
- 字符串格式CSV:比PNG慢。浮点数据转字符串的CPU开销大,且文本格式IO体积远大于压缩后的PNG,增加磁盘写入时间。
- 二进制格式文件:性能接近或优于PNG。直接写入原始浮点数据,体积与未压缩纹理相当,且避免了PNG的压缩开销。
实现示例(二进制写入优化版)
private BinaryWriter binaryWriter; protected void Start() { // ...其他初始化 string binaryPath = Path.Combine(customOutputFolderPath, "optical_flow.bin"); binaryWriter = new BinaryWriter(File.Open(binaryPath, FileMode.Create)); // 写入元数据:宽、高 binaryWriter.Write(targetTextureWidth); binaryWriter.Write(targetTextureHeight); } void OnCompleteReadback(AsyncGPUReadbackRequest request) { if (request.hasError) return; var data = request.GetData<float>(); // 写入帧索引 binaryWriter.Write(imageCount); // 写入所有像素的FlowX、FlowY(取RG通道) for (int i = 0; i < data.Length; i += 4) { binaryWriter.Write(data[i]); // FlowX binaryWriter.Write(data[i + 1]); // FlowY } // 异步刷新缓冲区 binaryWriter.FlushAsync(); imageCount++; } protected void OnDestroy() { // ...其他资源释放 binaryWriter?.Close(); binaryWriter?.Dispose(); }
内容的提问来源于stack exchange,提问作者Can Celik
相关产品推荐
相关产品推荐

