You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

CameraX+ML Kit实时文字识别预览框错位问题求助

CameraX + ML Kit 实时文字识别:Bounding Box 偏移问题

我正在开发一款Android应用,使用CameraX和ML Kit实现实时文字识别功能,当前遇到以下情况:

  • 相机预览可正常显示实时画面
  • ML Kit文字识别可处理图像并识别文本块
  • 系统会为检测到的文本元素绘制bounding box
  • 但这些框体无法与预览中的实际文字对齐,存在偏移或缩放错误

相关代码片段

Main Activity 代码片段

package com.aviskaarlab.booksnap.ui.views.home

import android.Manifest
import android.content.pm.PackageManager
import android.os.Build
import android.os.Bundle
import android.util.Log
import androidx.appcompat.app.AppCompatActivity
import androidx.camera.core.AspectRatio
import androidx.camera.core.CameraSelector
import androidx.camera.core.ImageAnalysis
import androidx.camera.core.Preview
import androidx.camera.core.resolutionselector.AspectRatioStrategy
import androidx.camera.core.resolutionselector.ResolutionSelector
import androidx.camera.lifecycle.ProcessCameraProvider
import androidx.camera.view.PreviewView
import androidx.core.content.ContextCompat
import java.util.concurrent.ExecutorService
import java.util.concurrent.Executors

class MainActivity : AppCompatActivity() {
    private lateinit var viewBinding: ActivityMainBinding
    private lateinit var cameraExecutor: ExecutorService
    private lateinit var textOverlay: TextOverlay
    private lateinit var viewFinder: PreviewView

    override fun onCreate(savedInstanceState: Bundle?) {
        super.onCreate(savedInstanceState)
        viewBinding = ActivityMainBinding.inflate(layoutInflater)
        setContentView(viewBinding.root)

        viewFinder = viewBinding.previewView
        textOverlay = viewBinding.textOverlay

        // Initialize camera and start preview
        startCamera()

        cameraExecutor = Executors.newSingleThreadExecutor()
    }

    private fun startCamera() {
        val cameraProviderFuture = ProcessCameraProvider.getInstance(this)

        cameraProviderFuture.addListener({
            val resolutionSelector = ResolutionSelector.Builder()
                .setAspectRatioStrategy(AspectRatioStrategy.RATIO_4_3_FALLBACK_AUTO_STRATEGY)
                .build()

            val rotation = viewFinder.display.rotation
            val cameraProvider: ProcessCameraProvider = cameraProviderFuture.get()

            // Preview
            val preview = Preview.Builder()
                .setResolutionSelector(resolutionSelector)
                .setTargetRotation(rotation)
                .build()
                .also {
                    it.setSurfaceProvider(viewBinding.previewView.surfaceProvider)
                }

            // Image Analysis
            val imageAnalysis = ImageAnalysis.Builder()
                .setResolutionSelector(resolutionSelector)
                .setTargetRotation(rotation)
                .build()
                .also {
                    it.setAnalyzer(
                        cameraExecutor,
                        BookSnapWordAnalyzer(
                            textOverlay,
                            viewBinding.previewView,
                        )
                    )
                }

            // Select back camera as default
            val cameraSelector = CameraSelector.DEFAULT_BACK_CAMERA

            try {
                cameraProvider.unbindAll()
                cameraProvider.bindToLifecycle(
                    this, cameraSelector, preview, imageAnalysis
                )

                preview.setSurfaceProvider(viewFinder.surfaceProvider)
            } catch (exc: Exception) {
                Log.e(TAG, "Use case binding failed", exc)
            }

        }, ContextCompat.getMainExecutor(this))
    }

    override fun onDestroy() {
        super.onDestroy()
        cameraExecutor.shutdown()
    }

    companion object {
        private const val TAG = "CameraXApp"
    }
}

BookSnapWordAnalyzer 代码片段

package com.aviskaarlab.booksnap.ui.views.home

import android.graphics.Matrix
import android.graphics.Rect
import android.graphics.RectF
import android.util.Log
import androidx.camera.core.ExperimentalGetImage
import androidx.camera.core.ImageAnalysis
import androidx.camera.core.ImageProxy
import androidx.camera.view.PreviewView
import com.google.mlkit.vision.common.InputImage
import com.google.mlkit.vision.text.Text
import com.google.mlkit.vision.text.TextRecognition
import com.google.mlkit.vision.text.latin.TextRecognizerOptions

internal class BookSnapWordAnalyzer(
    private val overlay: TextOverlay,
    private val previewView: PreviewView,
) : ImageAnalysis.Analyzer {

    companion object {
        private const val TAG = "BookSnapWordAnalyzer"
        private const val WORD_LENGTH = 4
    }

    private val recognizer = TextRecognition.getClient(TextRecognizerOptions.DEFAULT_OPTIONS)
    private lateinit var visionText: Text
    private lateinit var matrix: Matrix
    private var rotationDegrees: Int = 0

    @OptIn(ExperimentalGetImage::class)
    override fun analyze(imageProxy: ImageProxy) {
        val mediaImage = imageProxy.image ?: return
        rotationDegrees = imageProxy.imageInfo.rotationDegrees
        matrix = getCorrectionMatrix(imageProxy, previewView)

        val image =
            InputImage.fromMediaImage(mediaImage, imageProxy.imageInfo.rotationDegrees)

        recognizer.process(image)
            .addOnSuccessListener { visionText ->
                this.visionText = visionText
                val boxes = mutableListOf<CustomRect>()
                for (block in visionText.textBlocks) {
                    for (line in block.lines) {
                        for (element in line.elements) {
                            val elementText = element.text
                            val boundingBox = element.boundingBox
                            if (elementText.length >= WORD_LENGTH && boundingBox != null) {
                                boxes.add(
                                    CustomRect(
                                        adjustBoundingBox(boundingBox, imageProxy, previewView),
                                        elementText
                                    )
                                )
                            }
                        }
                    }
                }
                overlay.updateBoundingBoxes(boxes)
            }
            .addOnFailureListener { e ->
                Log.e(TAG, "Text recognition failed", e)
            }.addOnCompleteListener {
                imageProxy.close()
            }
    }

    private fun adjustBoundingBox(
        rect: Rect,
        imageProxy: ImageProxy,
        previewView: PreviewView
    ): RectF {
        val cropRect = imageProxy.cropRect
        val imageWidth = cropRect.width()
        val imageHeight = cropRect.height()

        val previewWidth = previewView.width
        val previewHeight = previewView.height

        val scaleX = previewWidth.toFloat() / imageWidth
        val scaleY = previewHeight.toFloat() / imageHeight

        val verticalOffset = (previewHeight - imageHeight * scaleY) / 2
        val horizontalOffset = (previewWidth - imageWidth * scaleX) / 2

        val left = rect.left * scaleX + horizontalOffset
        val top = rect.top * scaleY + verticalOffset
        val right = rect.right * scaleX + horizontalOffset
        val bottom = rect.bottom * scaleY + verticalOffset

        return RectF(left, top, right, bottom)
    }

    private fun getCorrectionMatrix(
        imageProxy: ImageProxy,
        previewView: PreviewView,
    ): Matrix {
        val cropRect = imageProxy.cropRect
        val matrix = Matrix()

        val source = floatArrayOf(
            cropRect.left.toFloat(), cropRect.top.toFloat(),
            cropRect.right.toFloat(), cropRect.top.toFloat(),
            cropRect.right.toFloat(), cropRect.bottom.toFloat(),
            cropRect.left.toFloat(), cropRect.bottom.toFloat()
        )

        val destination = floatArrayOf(
            0f, 0f,
            previewView.width.toFloat(), 0f,
            previewView.width.toFloat(), previewView.height.toFloat(),
            0f, previewView.height.toFloat()
        )

        matrix.setPolyToPoly(source, 0, destination, 0, 4)
        return matrix
    }
}

TextOverlay 代码片段

package com.aviskaarlab.booksnap.ui.views.home

import android.content.Context
import android.graphics.Canvas
import android.graphics.Color
import android.graphics.Paint
import android.graphics.RectF
import android.util.AttributeSet
import android.view.MotionEvent
import android.view.View

class TextOverlay @JvmOverloads constructor(
    context: Context, attrs: AttributeSet? = null, defStyleAttr: Int = 0
) : View(context, attrs, defStyleAttr) {

    private val paint = Paint().apply {
        color = Color.RED
        style = Paint.Style.STROKE
        strokeWidth = 2.0f
    }

    private val boundingBoxes = mutableListOf<CustomRect>()
    private var clickListener: ((String) -> Unit)? = null

    fun setOnRectangleClickListener(listener: (String) -> Unit) {
        clickListener = listener
    }

    fun updateBoundingBoxes(newBoundingBoxes: List<CustomRect>) {
        boundingBoxes.clear()
        boundingBoxes.addAll(newBoundingBoxes)
        invalidate() // Redraw the view
    }

    override fun onDraw(canvas: Canvas) {
        super.onDraw(canvas)
        for (box in boundingBoxes) {
            canvas.drawRect(box.rect, paint)
        }
    }

    override fun onTouchEvent(event: MotionEvent): Boolean {
        if (event.action == MotionEvent.ACTION_UP) {
            val x = event.x
            val y = event.y
            for (box in boundingBoxes) {
                if (box.rect.contains(x, y)) {
                    clickListener?.invoke("Clicked on rectangle ${box.text}")
                    return true
                }
            }
        }
        return true // Event handled
    }
}

解决方案:修正Bounding Box坐标对齐问题

问题根源在于当前adjustBoundingBox方法未正确处理CameraX图像旋转、PreviewView缩放模式以及图像分析帧与预览帧的坐标映射关系,以下是具体修正步骤:

1. 使用CameraX官方工具类处理坐标映射

CameraX提供CoordinateTransform工具类,可直接将图像分析帧的坐标转换为PreviewView的视图坐标,无需手动计算缩放和偏移:

步骤1:添加依赖

确保build.gradle中包含CameraX View依赖:

implementation "androidx.camera:camera-view:1.3.0"

步骤2:替换adjustBoundingBox方法

修改BookSnapWordAnalyzer中的adjustBoundingBox方法:

import androidx.camera.view.CoordinateTransform

private fun adjustBoundingBox(rect: Rect, imageProxy: ImageProxy): RectF {
    // 将MLKit返回的Rect转换为ImageProxy坐标系下的RectF
    val imageRect = RectF(rect)
    
    // 创建坐标转换器,自动处理旋转、缩放和偏移
    val transform = CoordinateTransform(
        imageProxy.imageInfo,
        previewView.display,
        previewView.scaleType
    )
    
    // 将图像坐标系的Rect映射到PreviewView的视图坐标系
    transform.mapRectToView(imageRect)
    
    return imageRect
}

2. 确保旋转角度处理正确

MLKit返回的bounding box基于旋转后的图像,需保证InputImage使用的旋转角度与ImageProxy一致:

// 保持原InputImage创建代码即可,无需修改
val image = InputImage.fromMediaImage(
    mediaImage,
    imageProxy.imageInfo.rotationDegrees
)

3. 统一PreviewView缩放模式

在布局文件中设置PreviewView的缩放类型,并保持代码中无冲突设置:

<!-- activity_main.xml 中的PreviewView -->
<androidx.camera.view.PreviewView
    android:id="@+id/previewView"
    android:layout_width="match_parent"
    android:layout_height="match_parent"
    app:scaleType="fillCenter" />

4. 补充缺失的CustomRect类

代码中用到的CustomRect未定义,需在同一包下添加:

data class CustomRect(val rect: RectF, val text: String)

5. 验证修改

替换代码后重新运行应用,此时bounding box应能准确覆盖预览中的文字。若仍有偏移,检查:

  • PreviewView的缩放类型在布局和代码中是否一致
  • ImageAnalysis和Preview是否使用了相同的ResolutionSelector
  • 设备屏幕旋转是否被正确处理

内容的提问来源于stack exchange,提问作者Purushotam Kumar

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 01:55:55