You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Compose Multiplatform iOS端Vision OCR性能问题排查求助

iOS端Compose Multiplatform摄像头文字识别性能问题排查与修复

问题根源分析

你的代码存在几个核心问题,直接导致iOS端性能崩溃、UI无响应:

  1. 主线程完全阻塞:setSampleBufferDelegate将帧回调绑定到主队列,且Vision文字识别同步在主队列执行,摄像头每帧回调都占用主线程,导致UI彻底无法响应。
  2. 重复创建识别请求:每次处理帧都新建VNRecognizeTextRequest,不必要的对象创建加剧了性能消耗。
  3. 无限制帧处理:摄像头每秒输出30-60帧,每帧都执行文字识别,计算量远超设备负荷,触发资源耗尽错误。
  4. UI更新线程错误:识别回调未切回主线程,导致Compose状态更新失效,即使日志能识别到文字,UI也不刷新。

针对性修复方案

1. 创建专用串行队列处理帧与识别

为摄像头输出和Vision识别单独创建串行队列,彻底与主线程解耦:

import platform.darwin.dispatch_queue_create
import platform.darwin.DISPATCH_QUEUE_SERIAL

// 在类中定义专用队列
private val captureQueue = dispatch_queue_create("com.yourapp.camera.queue", DISPATCH_QUEUE_SERIAL)

2. 复用识别请求,避免重复创建

初始化时只创建一次VNRecognizeTextRequest,并启用快速识别模式降低开销:

// 初始化时创建请求
textRecognitionRequest = VNRecognizeTextRequest { request, error ->
    if (error != null) {
        println("CameraPermissionManager || Error recognizing text: $error")
        return@VNRecognizeTextRequest
    }
    val observations = request?.results?.filterIsInstance<VNRecognizedTextObservation>()
    observations?.firstOrNull()?.let { observation ->
        val topText = observation.topCandidates(1u).firstOrNull()?.string ?: ""
        // 切回主线程更新UI
        dispatch_async(dispatch_get_main_queue()) {
            onTextDetected(topText)
        }
    }
}.apply {
    usesLanguageCorrection = false
    recognitionLevel = VNRequestTextRecognitionLevelFast
}

3. 限制识别频率

添加时间戳判断,控制每秒处理1-2帧,减少计算压力:

private var lastProcessTime = 0L

private fun processSampleBuffer(
    sampleBuffer: CMSampleBufferRef,
    onTextDetected: (String) -> Unit
) {
    val currentTime = System.currentTimeMillis()
    // 跳过间隔过短的帧
    if (currentTime - lastProcessTime < 500) return
    lastProcessTime = currentTime

    val pixelBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) ?: run {
        println("CameraPermissionManager || Pixel buffer is null")
        return
    }

    val handler = VNImageRequestHandler(pixelBuffer, options = emptyMap())
    try {
        handler.performRequests(listOf(textRecognitionRequest), null)
    } catch (e: Exception) {
        println("CameraPermissionManager || Error performing Vision request: $e")
    }
}

4. 修正SampleBufferDelegate的队列设置

将帧回调切换到专用队列,避免阻塞主线程:

val videoOutput = AVCaptureVideoDataOutput().apply {
    videoSettings = mapOf(kCVPixelBufferPixelFormatTypeKey to kCVPixelFormatType_32BGRA)
    setSampleBufferDelegate(
        createSampleBufferDelegate(onTextDetected),
        captureQueue // 使用专用队列而非主队列
    )
    alwaysDiscardsLateVideoFrames = true // 丢弃延迟帧,避免队列堆积
}

完整优化后的代码

import androidx.compose.foundation.layout.fillMaxSize
import androidx.compose.runtime.Composable
import androidx.compose.runtime.DisposableEffect
import androidx.compose.runtime.LaunchedEffect
import androidx.compose.runtime.getValue
import androidx.compose.runtime.mutableStateOf
import androidx.compose.runtime.remember
import androidx.compose.runtime.setValue
import androidx.compose.ui.Modifier
import androidx.compose.ui.interop.UIKitView
import kotlinx.cinterop.ExperimentalForeignApi
import kotlinx.cinterop.useContents
import platform.AVFoundation.*
import platform.CoreGraphics.CGRectMake
import platform.CoreMedia.CMSampleBufferGetImageBuffer
import platform.CoreMedia.CMSampleBufferRef
import platform.CoreVideo.kCVPixelBufferPixelFormatTypeKey
import platform.CoreVideo.kCVPixelFormatType_32BGRA
import platform.UIKit.UIScreen
import platform.UIKit.UIView
import platform.Vision.*
import platform.darwin.*

@OptIn(ExperimentalForeignApi::class)
actual class CameraPermissionManager {

    private lateinit var captureSession: AVCaptureSession
    private lateinit var textRecognitionRequest: VNRecognizeTextRequest
    private lateinit var videoLayer: AVCaptureVideoPreviewLayer
    private val captureQueue = dispatch_queue_create("com.yourapp.camera.queue", DISPATCH_QUEUE_SERIAL)
    private var lastProcessTime = 0L


    @Composable
    actual fun RequestCameraPermission(
        onPermissionGranted: @Composable () -> Unit,
        onPermissionDenied: @Composable () -> Unit
    ) {
        var hasPermission by remember { mutableStateOf(false) }
        var permissionRequested by remember { mutableStateOf(false) }

        LaunchedEffect(Unit) {
            val status = AVCaptureDevice.authorizationStatusForMediaType(AVMediaTypeVideo)
            when (status) {
                AVAuthorizationStatusAuthorized -> {
                    hasPermission = true
                }

                AVAuthorizationStatusNotDetermined -> {
                    AVCaptureDevice.requestAccessForMediaType(AVMediaTypeVideo) { granted ->
                        dispatch_async(dispatch_get_main_queue()) {
                            hasPermission = granted
                            permissionRequested = true
                        }
                    }
                }

                AVAuthorizationStatusDenied, AVAuthorizationStatusRestricted -> {
                    hasPermission = false
                }

                else -> {}
            }
        }

        if (hasPermission) {
            onPermissionGranted()
        } else if (permissionRequested) {
            onPermissionDenied()
        }
    }

    @Composable
    actual fun StartCameraPreview(onTextDetected: (String) -> Unit) {
        val screenWidth = UIScreen.mainScreen.bounds.useContents { size.width }
        val screenHeight = UIScreen.mainScreen.bounds.useContents { size.height }
        val previewView = remember {
            UIView(frame = CGRectMake(0.0, 0.0, screenWidth, screenHeight))
        }

        LaunchedEffect(Unit) {
            // Setup AVCaptureSession
            captureSession = AVCaptureSession().apply {
                sessionPreset = AVCaptureSessionPresetPhoto
            }

            val device = AVCaptureDevice.defaultDeviceWithMediaType(AVMediaTypeVideo)
            val input = device?.let { AVCaptureDeviceInput.deviceInputWithDevice(it, null) }
            if (input != null && captureSession.canAddInput(input)) {
                captureSession.addInput(input)
            }

            videoLayer = AVCaptureVideoPreviewLayer(session = captureSession).apply {
                this.videoGravity = AVLayerVideoGravityResizeAspectFill
                this.frame = previewView.bounds
            }
            previewView.layer.addSublayer(videoLayer)

            // Setup Vision Text Recognition (复用请求)
            textRecognitionRequest = VNRecognizeTextRequest { request, error ->
                if (error != null) {
                    println("CameraPermissionManager || Error recognizing text: $error")
                    return@VNRecognizeTextRequest
                }
                val observations = request?.results?.filterIsInstance<VNRecognizedTextObservation>()
                observations?.firstOrNull()?.let { observation ->
                    val topText = observation.topCandidates(1u).firstOrNull()?.string ?: ""
                    // 必须切回主线程更新Compose状态
                    dispatch_async(dispatch_get_main_queue()) {
                        onTextDetected(topText)
                    }
                }
            }.apply {
                // 启用快速识别模式,降低计算开销
                usesLanguageCorrection = false
                recognitionLevel = VNRequestTextRecognitionLevelFast
            }

            val videoOutput = AVCaptureVideoDataOutput().apply {
                videoSettings = mapOf(kCVPixelBufferPixelFormatTypeKey to kCVPixelFormatType_32BGRA)
                // 使用专用队列处理帧回调
                setSampleBufferDelegate(
                    createSampleBufferDelegate(onTextDetected),
                    captureQueue
                )
                // 丢弃延迟帧,避免队列堆积
                alwaysDiscardsLateVideoFrames = true
            }

            if (captureSession.canAddOutput(videoOutput)) {
                captureSession.addOutput(videoOutput)
                // 调整连接方向,避免预览画面旋转
                videoOutput.connectionWithMediaType(AVMediaTypeVideo)?.apply {
                    videoOrientation = AVCaptureVideoOrientation.Portrait
                }
            }

            captureSession.startRunning()
        }

        DisposableEffect(Unit) {
            onDispose {
                if (::captureSession.isInitialized) {
                    captureSession.stopRunning()
                }
            }
        }

        UIKitView(
            factory = { previewView },
            modifier = Modifier.fillMaxSize()
        )
    }

    private fun createSampleBufferDelegate(onTextDetected: (String) -> Unit): AVCaptureVideoDataOutputSampleBufferDelegateProtocol {
        return object : NSObject(), AVCaptureVideoDataOutputSampleBufferDelegateProtocol {
            override fun captureOutput(
                output: AVCaptureOutput,
                didOutputSampleBuffer: CMSampleBufferRef?,
                fromConnection: AVCaptureConnection
            ) {
                didOutputSampleBuffer?.let {
                    processSampleBuffer(it, onTextDetected)
                }
            }
        }
    }

    private fun processSampleBuffer(
        sampleBuffer: CMSampleBufferRef,
        onTextDetected: (String) -> Unit
    ) {
        val currentTime = System.currentTimeMillis()
        // 限制识别频率,每500ms处理一次
        if (currentTime - lastProcessTime < 500) return
        lastProcessTime = currentTime

        val pixelBuffer = CMSampleBufferGetImageBuffer(sampleBuffer) ?: run {
            println("CameraPermissionManager || Pixel buffer is null")
            return
        }

        val handler = VNImageRequestHandler(pixelBuffer, options = emptyMap())
        try {
            handler.performRequests(listOf(textRecognitionRequest), null)
        } catch (e: Exception) {
            println("CameraPermissionManager || Error performing Vision request: $e")
        }
    }
}

额外优化建议

  • 降低摄像头分辨率:将sessionPreset改为AVCaptureSessionPresetMedium,减少每帧数据量。
  • 缩小识别范围:设置textRecognitionRequest.regionOfInterest,只识别画面中指定区域。
  • 动态调整识别频率:根据设备性能动态调整帧间隔,避免低端设备过载。

内容的提问来源于stack exchange,提问作者DoctorWho

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.18 15:08:13