AVCaptureDepthDataOutput数据更新延迟问题排查求助
问题:深度数据回调频率远低于视频数据
我有一个同时实现AVCaptureVideoDataOutputSampleBufferDelegate和AVCaptureDepthDataOutputDelegate协议的ViewController,需要同时采集视频数据(用于Vision ML推理)和深度数据(计算相机到特定点的距离)。但目前depthDataOutput的触发频率远低于captureOutput,深度数据更新过慢,求原因及解决办法。
相关实现代码
视频数据回调扩展
extension MainRecognizerViewController: AVCaptureVideoDataOutputSampleBufferDelegate { func captureOutput(_ output: AVCaptureOutput, didOutput sampleBuffer: CMSampleBuffer, from connection: AVCaptureConnection) { DispatchQueue.main.async { self.captureSessionManager.manageFlashlight(for: sampleBuffer, force: nil) } guard let cvPixelBuffer = sampleBuffer.convertToPixelBuffer() else { return } let exifOrientation = exifOrientationFromDeviceOrientation() let handler = VNImageRequestHandler(cvPixelBuffer: cvPixelBuffer, orientation: exifOrientation) let objectsRecognitionRequest = prepareVisionRequestForObjectsRecognition( pixelBuffer: cvPixelBuffer ) DispatchQueue.global().async { try? handler.perform([objectsRecognitionRequest]) try? handler.perform(self.roadLightsRecognizerRequests) try? handler.perform(self.pedestrianCrossingRecognizerRequests) } } }
深度数据回调扩展
extension MainRecognizerViewController: AVCaptureDepthDataOutputDelegate { func depthDataOutput(_ output: AVCaptureDepthDataOutput, didOutput depthData: AVDepthData, timestamp: CMTime, connection: AVCaptureConnection) { if depthMeasurementsLeftInLoop == 0 { depthMeasurementsCumul = 0.0 depthMeasurementMin = 9999.9 depthMeasurementMax = 0.0 depthMeasurementsLeftInLoop = depthMeasurementRepeats } if depthMeasurementsLeftInLoop > 0 { var convertedDepthData: AVDepthData = depthData.converting( toDepthDataType: kCVPixelFormatType_DepthFloat16 ) let depthFrame = convertedDepthData.depthDataMap let depthPoint = CGPoint(x: CGFloat(CVPixelBufferGetWidth(depthFrame)) / 2, y: CGFloat(CVPixelBufferGetHeight(depthFrame) / 2)) let depthVal = getDepthValueFromFrame(fromFrame: depthFrame, atPoint: depthPoint) print(depthVal) let measurement = depthVal * 100 depthMeasurementsCumul += measurement if measurement > depthMeasurementMax { depthMeasurementMax = measurement } if measurement < depthMeasurementMin { depthMeasurementMin = measurement } depthMeasurementsLeftInLoop -= 1 DispatchQueue.main.async { [weak self] in self?.distanceMeasurerViewModel?.distanceString = String(format: "%.2f", measurement) } } } }
捕获会话管理类
import AVFoundation final class CaptureSessionManager: CaptureSessionManaging { @Inject private var flashlightManager: FlashlightManaging private let captureSessionQueue = DispatchQueue(label: "captureSessionQueue") private let captureSessionDataOutputQueue = DispatchQueue( label: "captureSessionVideoDataOutput", qos: .userInitiated, attributes: [], autoreleaseFrequency: .workItem ) private var sampleBufferOutput: AVCaptureVideoDataOutput = AVCaptureVideoDataOutput() private var sampleBufferDelegate: AVCaptureVideoDataOutputSampleBufferDelegate? private var depthDataOutput: AVCaptureDepthDataOutput = AVCaptureDepthDataOutput() private var depthDataOutputDelegate: AVCaptureDepthDataOutputDelegate? var cameraMode: CameraMode? private var desiredFrameRate: Double? private var videoDevice: AVCaptureDevice? = AVCaptureDevice.default( .builtInLiDARDepthCamera, for: .video, position: .back ) var bufferSize: CGSize = .zero var captureSession: AVCaptureSession! func setUp(with sampleBufferDelegate: AVCaptureVideoDataOutputSampleBufferDelegate, and depthDataOutputDelegate: AVCaptureDepthDataOutputDelegate, for cameraMode: CameraMode, cameraPosition: AVCaptureDevice.Position, desiredFrameRate: Double, completion: @escaping () -> ()) { stopCaptureSession() self.sampleBufferDelegate = sampleBufferDelegate self.depthDataOutputDelegate = depthDataOutputDelegate self.cameraMode = cameraMode self.desiredFrameRate = desiredFrameRate authorizeCaptureSession { completion() } } func manageFlashlight(for sampleBuffer: CMSampleBuffer?, force torchMode: AVCaptureDevice.TorchMode?) { flashlightManager.manageFlashlight(for: sampleBuffer, and: self.videoDevice, force: torchMode) } private func authorizeCaptureSession(completion: @escaping () -> ()) { switch AVCaptureDevice.authorizationStatus(for: .video) { case .authorized: setupCaptureSession { completion() } case .notDetermined: AVCaptureDevice.requestAccess(for: .video) { [weak self] granted in if granted { self?.setupCaptureSession { completion() } } } default: return } } private func setupCaptureSession(completion: @escaping () -> ()) { captureSessionQueue.async { [unowned self] in var captureSession: AVCaptureSession = AVCaptureSession() captureSession.beginConfiguration() guard let videoDevice = videoDevice else { return } do { let captureDeviceInput = try AVCaptureDeviceInput(device: videoDevice) guard captureSession.canAddInput(captureDeviceInput) else { return } captureSession.addInput(captureDeviceInput) } catch { return } let sessionPreset: SessionPreset = .hd1280x720 guard let videoSetupedCaptureSession: AVCaptureSession = setupCaptureSessionForVideo( captureSession: captureSession, sessionPreset: sessionPreset ) else { return } guard let depthAndVideoSetupedCaptureSession = setupCaptureSessionForDepth( captureSession: videoSetupedCaptureSession ) else { return } captureSession = depthAndVideoSetupedCaptureSession captureSession.sessionPreset = sessionPreset.preset captureSession.commitConfiguration() self.captureSession = captureSession self.startCaptureSession() completion() } } private func setupCaptureSessionForVideo(captureSession: AVCaptureSession, sessionPreset: SessionPreset) -> AVCaptureSession? { let captureSessionVideoOutput: AVCaptureVideoDataOutput = AVCaptureVideoDataOutput() captureSessionVideoOutput.videoSettings = [ kCVPixelBufferPixelFormatTypeKey as String: NSNumber( value: kCMPixelFormat_32BGRA ) ] captureSessionVideoOutput.alwaysDiscardsLateVideoFrames = true captureSessionVideoOutput.setSampleBufferDelegate( self.sampleBufferDelegate, queue: captureSessionDataOutputQueue ) guard let videoDevice = videoDevice else { return nil } var formatToSet: AVCaptureDevice.Format = videoDevice.formats[0] guard let desiredFrameRate = desiredFrameRate else { return nil } for format in videoDevice.formats.reversed() { let ranges = format.videoSupportedFrameRateRanges let frameRates = ranges[0] if desiredFrameRate <= frameRates.maxFrameRate, format.formatDescription.dimensions.width == sessionPreset.formatWidth, format.formatDescription.dimensions.height == sessionPreset.formatHeight { formatToSet = format break } } do { try videoDevice.lockForConfiguration() if videoDevice.hasTorch { self.manageFlashlight(for: nil, force: .auto) } let dimensions = CMVideoFormatDescriptionGetDimensions((videoDevice.activeFormat.formatDescription)) bufferSize.width = CGFloat(dimensions.width) bufferSize.height = CGFloat(dimensions.height) videoDevice.activeFormat = formatToSet let timescale = CMTimeScale(desiredFrameRate) if videoDevice.activeFormat.videoSupportedFrameRateRanges[0].maxFrameRate >= desiredFrameRate { videoDevice.activeVideoMinFrameDuration = CMTime(value: 1, timescale: timescale) videoDevice.activeVideoMaxFrameDuration = CMTime(value: 1, timescale: timescale) } videoDevice.unlockForConfiguration() } catch { return nil } guard captureSession.canAddOutput(captureSessionVideoOutput) else { return nil } let captureConnection = captureSessionVideoOutput.connection(with: .video) captureConnection?.isEnabled = true captureSession.addOutput(captureSessionVideoOutput) if let cameraMode = self.cameraMode, CameraMode.modesWithPortraitVideoConnection.contains(cameraMode) { captureSessionVideoOutput.connection(with: .video)?.videoOrientation = .portrait } return captureSession } private func setupCaptureSessionForDepth(captureSession: AVCaptureSession) -> AVCaptureSession? { guard let depthDataOutputDelegate = depthDataOutputDelegate else { return nil } if captureSession.canAddOutput(depthDataOutput) { captureSession.addOutput(depthDataOutput) depthDataOutput.isFilteringEnabled = false } else { return nil } if let connection = depthDataOutput.connection(with: .depthData) { connection.isEnabled = true depthDataOutput.isFilteringEnabled = false depthDataOutput.setDelegate( depthDataOutputDelegate, callbackQueue: captureSessionDataOutputQueue ) } else { return nil } guard let videoDevice = videoDevice else { return nil } let availableFormats = videoDevice.activeFormat.supportedDepthDataFormats let availableHdepFormats = availableFormats.filter { f in CMFormatDescriptionGetMediaSubType(f.formatDescription) == kCVPixelFormatType_DepthFloat16 } let selectedFormat = availableHdepFormats.max(by: { lower, higher in CMVideoFormatDescriptionGetDimensions(lower.formatDescription).width < CMVideoFormatDescriptionGetDimensions(higher.formatDescription).width }) do { try videoDevice.lockForConfiguration() videoDevice.activeDepthDataFormat = selectedFormat videoDevice.unlockForConfiguration() } catch { return nil } return captureSession } func startCaptureSession() { self.captureSession?.startRunning() } func stopCaptureSession() { self.captureSession?.stopRunning() } }
可能的原因及解决方案
1. 视频与深度格式的帧率不匹配
原因:你强制设置了视频的固定帧率,但所选视频格式对应的深度数据输出帧率可能低于视频帧率(比如部分设备在1080p/30fps视频下,深度帧率仅支持15fps)。
解决方案:修改视频格式选择逻辑,优先选择深度数据能跟上目标帧率的格式:
for format in videoDevice.formats.reversed() { let ranges = format.videoSupportedFrameRateRanges let frameRates = ranges[0] // 检查该视频格式对应的深度数据支持的帧率 let depthFormats = format.supportedDepthDataFormats guard !depthFormats.isEmpty else { continue } let depthFrameRateRange = depthFormats.first!.videoSupportedFrameRateRanges.first! if desiredFrameRate <= frameRates.maxFrameRate, desiredFrameRate <= depthFrameRateRange.maxFrameRate, format.formatDescription.dimensions.width == sessionPreset.formatWidth, format.formatDescription.dimensions.height == sessionPreset.formatHeight { formatToSet = format break } }
2. 回调队列阻塞
原因:视频和深度数据共享同一个回调队列,视频回调中的Vision推理耗时较长,会阻塞队列,导致深度回调被延迟处理。
解决方案:为深度数据分配独立的回调队列:
// 在CaptureSessionManager中新增深度队列 private let captureSessionDepthDataOutputQueue = DispatchQueue( label: "captureSessionDepthDataOutput", qos: .userInitiated, attributes: [], autoreleaseFrequency: .workItem ) // 在setupCaptureSessionForDepth中替换队列 depthDataOutput.setDelegate( depthDataOutputDelegate, callbackQueue: captureSessionDepthDataOutputQueue )
3. 深度数据分辨率过高
原因:你选择了最高分辨率的深度格式,高分辨率深度数据的处理和输出开销更大,会降低帧率。
解决方案:如果不需要高分辨率,选择更低分辨率的深度格式:
// 选择最低分辨率的深度格式 let selectedFormat = availableHdepFormats.min(by: { lower, higher in CMVideoFormatDescriptionGetDimensions(lower.formatDescription).width < CMVideoFormatDescriptionGetDimensions(higher.formatDescription).width })
4. Vision推理占用过多资源
原因:多次调用handler.perform执行Vision请求,占用大量CPU/GPU资源,系统会自动降低深度数据输出帧率。
解决方案:合并Vision请求,减少执行次数:
DispatchQueue.global().async { let allRequests = [objectsRecognitionRequest] + self.roadLightsRecognizerRequests + self.pedestrianCrossingRecognizerRequests try? handler.perform(allRequests) }
内容的提问来源于stack exchange,提问作者Vader20FF
相关产品推荐
相关产品推荐

