基于人脸位置横屏转竖屏裁剪时导出会话失败问题
基于AVFoundation+Vision实现人脸跟踪动态裁剪横屏视频为竖屏的解决方案
1. 先定位ExportSession失败的核心原因
添加人脸跟踪指令后导出失败,大概率是以下几个细节没处理好:
- 检查
faceTrackInstructions的时间范围是否与视频片段的时间范围完全匹配,AVFoundation对时间精度要求极高,毫秒级偏差都会导致导出失败。 - 确认人脸坐标到视频渲染坐标的转换逻辑是否正确:Vision的坐标系是图像归一化坐标(原点左上,y轴向下),而AVFoundation的视频渲染坐标系会受视频方向影响,必须做翻转/平移修正。
- 避免使用固定分辨率的导出预设,优先用
AVAssetExportPresetHighestQuality,动态裁剪可能导致分辨率不兼容固定预设。 - 导出前必须调用
exportSession.canExport()验证,若返回false,直接打印exportSession.error?.userInfo,里面会有具体的错误原因(比如AVErrorInvalidVideoComposition通常是指令矩阵或时间范围错误)。
2. 正确实现人脸跟踪驱动的动态裁剪逻辑
步骤1:预分析视频帧,记录人脸轨迹
用Vision批量提取视频关键帧的人脸位置,保存时间戳对应的人脸 bounding box:
func getFaceTracks(from asset: AVAsset) async throws -> [CMTime: CGRect] { guard let videoTrack = asset.tracks(withMediaType: .video).first else { throw NSError(domain: "TrackError", code: -1, userInfo: [NSLocalizedDescriptionKey: "无视频轨道"]) } let reader = try AVAssetReader(asset: asset) let outputSettings = [kCVPixelBufferPixelFormatTypeKey as String: kCVPixelFormatType_32BGRA] let trackOutput = AVAssetReaderTrackOutput(track: videoTrack, outputSettings: outputSettings) reader.add(trackOutput) reader.startReading() var faceTimestamps = [CMTime: CGRect]() let faceRequest = VNDetectFaceRectanglesRequest() let videoNaturalSize = videoTrack.naturalSize.applying(videoTrack.preferredTransform).size while let sampleBuffer = trackOutput.copyNextSampleBuffer() { let timestamp = CMSampleBufferGetPresentationTimeStamp(sampleBuffer) guard let imageBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) else { continue } let handler = VNImageRequestHandler(cvPixelBuffer: imageBuffer, orientation: .up) try handler.perform([faceRequest]) if let faceObservation = faceRequest.results?.first { // 转换Vision归一化坐标到视频渲染坐标 let normalizedRect = faceObservation.boundingBox let faceRect = CGRect( x: normalizedRect.origin.x * videoNaturalSize.width, y: (1 - normalizedRect.origin.y - normalizedRect.height) * videoNaturalSize.height, width: normalizedRect.width * videoNaturalSize.width, height: normalizedRect.height * videoNaturalSize.height ) faceTimestamps[timestamp] = faceRect } } return faceTimestamps }
步骤2:构建动态裁剪的视频合成指令
基于人脸轨迹,为每个时间点生成裁剪变换矩阵,确保人脸始终处于竖屏画面中心:
func buildDynamicCropComposition(asset: AVAsset, faceTracks: [CMTime: CGRect]) -> AVMutableVideoComposition { guard let videoTrack = asset.tracks(withMediaType: .video).first else { return AVMutableVideoComposition() } let targetSize = CGSize(width: 1080, height: 1920) // 竖屏目标分辨率 let composition = AVMutableVideoComposition() composition.renderSize = targetSize composition.frameDuration = videoTrack.minFrameDuration composition.renderScale = 1.0 let layerInstruction = AVMutableVideoCompositionLayerInstruction(assetTrack: videoTrack) let mainInstruction = AVMutableVideoCompositionInstruction() mainInstruction.timeRange = CMTimeRange(start: .zero, duration: asset.duration) mainInstruction.layerInstructions = [layerInstruction] // 按时间戳排序,插入关键帧变换 let sortedTimestamps = faceTracks.keys.sorted { $0 < $1 } let videoSize = videoTrack.naturalSize.applying(videoTrack.preferredTransform).size for timestamp in sortedTimestamps { guard let faceRect = faceTracks[timestamp] else { continue } // 计算裁剪中心:以人脸中心为基准,限制在视频边界内 let cropCenterX = max(faceRect.midX, targetSize.width/2) let finalCenterX = min(cropCenterX, videoSize.width - targetSize.width/2) let cropCenterY = max(faceRect.midY, targetSize.height/2) let finalCenterY = min(cropCenterY, videoSize.height - targetSize.height/2) // 生成平移变换,将视频平移至人脸中心对齐裁剪区域中心 let transform = CGAffineTransform( translationX: -(finalCenterX - targetSize.width/2), y: -(finalCenterY - targetSize.height/2) ) layerInstruction.setTransform(transform, at: timestamp) // 为相邻时间点添加插值,保证裁剪过渡平滑 if let nextTimestamp = sortedTimestamps.first(where: { $0 > timestamp }) { let timeRange = CMTimeRange(start: timestamp, duration: CMTimeSubtract(nextTimestamp, timestamp)) layerInstruction.setTransformRamp(fromStart: transform, toEnd: transform, timeRange: timeRange) } } composition.instructions = [mainInstruction] return composition }
步骤3:执行导出操作
func exportVideo(asset: AVAsset, composition: AVMutableVideoComposition, outputURL: URL) async throws { let exportSession = AVAssetExportSession(asset: asset, presetName: AVAssetExportPresetHighestQuality)! exportSession.videoComposition = composition exportSession.outputURL = outputURL exportSession.outputFileType = .mp4 exportSession.shouldOptimizeForNetworkUse = true guard exportSession.canExport() else { throw exportSession.error ?? NSError(domain: "ExportError", code: -1, userInfo: [NSLocalizedDescriptionKey: "导出会话不可用"]) } try await withCheckedThrowingContinuation { continuation in exportSession.exportAsynchronously { switch exportSession.status { case .completed: continuation.resume() case .failed, .cancelled: let error = exportSession.error ?? NSError(domain: "ExportError", code: -1, userInfo: [NSLocalizedDescriptionKey: "导出失败或取消"]) continuation.resume(throwing: error) default: break } } } }
3. 关键注意事项
- 坐标系转换:一定要注意Vision与AVFoundation的坐标系差异,尤其是视频带有旋转方向时,需要结合
track.preferredTransform做修正。 - 边界限制:计算裁剪中心时必须限制在视频原始尺寸内,避免出现黑边或画面拉伸。
- 性能优化:逐帧分析人脸会拖慢速度,可改为每秒分析1-2帧,利用AVFoundation的变换插值实现平滑过渡。
- 错误排查:导出失败时一定要打印
exportSession.error的完整信息,这是定位问题最快的方式。
内容的提问来源于stack exchange,提问作者John
相关产品推荐
相关产品推荐

