You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于Python的数学公式图像解析:符号提取与定位优化问询

数学公式图片的符号分割与中心定位:标准方案探讨

我需要处理一张数学公式图片,解析其中的每个符号并保存每个符号的中心位置,最终将原图片转换为15张75×75的单符号图片。

我已尝试的方法:

  • 将图像转为灰度图后二值化:将像素值>250的接近白色区域设为255,其余设为0;
  • 使用BNF算法查找所有连通组件,再将组件转换为图像(包含缩放等操作)。

但我认为这并非最优方案,想了解该问题是否存在标准解决方案?

import cv2
import numpy as np
import math
from collections import deque

class Point:
    def __init__(self, x, y):
        self.x = x
        self.y = y

class RawImage:
    def __init__(self, image, center):
        self.image = image
        self.center = center

class Parser:

    def __init__(self, targetSizes=(75, 75), binaryThreshold=cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU,
                 scaleFully=False, scaleFullyRate=0.9, whiteThreshold=249, blackThreshold=0,
                 rescalingInterpolation=cv2.INTER_AREA, pixelsInImageThreshold=20,
                 rescaleOriginalImage=True, rescaleToAtLeast=200, rescaleToAtMaximum=1000):
        self.targetWidth = targetSizes[0]
        self.targetHeight = targetSizes[1]

        self.binaryThreshold = binaryThreshold

        self.scaleFully = scaleFully
        self.scaleFullyRate = scaleFullyRate

        self.whiteThreshold = whiteThreshold
        self.blackThreshold = blackThreshold

        self.rescalingInterpolation = rescalingInterpolation
        self.pixelsInImageThreshold = pixelsInImageThreshold

        self.rescaleOriginalImage = rescaleOriginalImage
        self.rescaleOriginalMin = rescaleToAtLeast
        self.rescaleOriginalMax = rescaleToAtMaximum

        self.parseMode = 1

    def _imageToBinary(self, image, zeroValueTrash=0, oneValueTrash=253):
        grayImage = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

        ret, binary = cv2.threshold(grayImage, self.blackThreshold, self.whiteThreshold, self.binaryThreshold)
        # cv2.imwrite("Test.png", binary)
        return binary

    def _BNF(self, binaryImage):

        Q = deque()
        whitePixels = []
        gg = 0
        for i in range(len(binaryImage)):
            for j in range(len(binaryImage[i])):
                if binaryImage[i][j] > self.whiteThreshold - 1:
                    Q.append((i, j))
                    binaryImage[i][j] = 0
                    obj = []
                    gg += 1
                    while Q:
                        i, j = Q.popleft()

                        obj.append((i, j))

                        if i + 1 < len(binaryImage) and binaryImage[i + 1][j] != 0:
                            Q.append((i + 1, j))
                            binaryImage[i + 1][j] = 0

                        if j - 1 > 0 and binaryImage[i][j - 1] != 0:
                            Q.append((i, j - 1))
                            binaryImage[i][j - 1] = 0

                        if i - 1 > 0 and binaryImage[i - 1][j] != 0:
                            Q.append((i - 1, j))
                            binaryImage[i - 1][j] = 0

                        if j + 1 < len(binaryImage[i]) and binaryImage[i][j + 1] != 0:
                            Q.append((i, j + 1))
                            binaryImage[i][j + 1] = 0

                        if self.parseMode == 1:
                            if i + 1 < len(binaryImage) and j + 1 < len(binaryImage[i+1]) and binaryImage[i + 1][j+1] != 0:
                                Q.append((i + 1, j+1))
                                binaryImage[i + 1][j+1] = 0
                            if i + 1 < len(binaryImage) and j - 1 > 0 and binaryImage[i + 1][j-1] != 0:
                                Q.append((i + 1, j-1))
                                binaryImage[i + 1][j-1] = 0
                            if i - 1 > 0 and j - 1 > 0 and binaryImage[i - 1][j - 1] != 0:
                                Q.append((i - 1, j - 1))
                                binaryImage[i - 1][j - 1] = 0
                            if i - 1 > 0 and j + 1 < len(binaryImage[i-1]) and binaryImage[i - 1][j + 1] != 0:
                                Q.append((i - 1, j + 1))
                                binaryImage[i - 1][j + 1] = 0

                    cv2.imwrite("tmp/{}.png".format(gg), binaryImage)
                    whitePixels.append(obj)
        return whitePixels

    def parseImage(self, image_path: str) -> list:

        image = cv2.imread(image_path)
        if self.rescaleOriginalImage:
            image = self.scaleOriginal(image)

        binary = self._imageToBinary(image)

        whitePixels = self._BNF(binary)

        return whitePixels

    def isScaleable(self, imageShape):
        return True

    def scaleOriginal(self, image: np.ndarray):
        # To be created
        return image

    @staticmethod
    def _getImageAndCenterFromDotes(Dotes, originalImage=None):
        i_mx, j_mx = -1, -1
        i_mn, j_mn = 100500, 100500  # 初始化为足够大的数值

        # 查找符号区域的边界
        for el in Dotes:
            i, j = el

            if i_mx < i:
                i_mx = i
            if j_mx < j:
                j_mx = j
            if j_mn > j:
                j_mn = j
            if i_mn > i:
                i_mn = i

        # 计算符号中心
        imageCenter = Point((i_mx + i_mn) // 2, (j_mx + j_mn) // 2)

        # 确定符号区域尺寸
        width, height = i_mx - i_mn + 1, j_mx - j_mn + 1
        image = np.zeros((width, height)) if originalImage is None else np.zeros((width, height, 3))

        # 从像素点重建符号图像
        if originalImage is not None:
            for el in Dotes:
                i, j = el
                image[i - i_mn][j - j_mn] = originalImage[i][j]
        else:
            for el in Dotes:
                i, j = el
                image[i - i_mn][j - j_mn] = 255

        return image, imageCenter

    def scaleParsedImage(self, image: np.ndarray):
        """
        将符号图像缩放到目标尺寸75×75,并居中放置
        :param image: np.ndarray
        :return: scaledImage np.ndarray
        """
        width, height = image.shape if len(image.shape) == 2 else image.shape[0], image.shape[1]

        newWidth = self.targetWidth if width > self.targetHeight else width
        newHeight = self.targetHeight if height > self.targetHeight else height
        if self.scaleFully and newHeight < self.targetHeight * self.scaleFullyRate and newWidth * self.scaleFullyRate:
            scaleRate = min((self.targetWidth * self.scaleFullyRate / newWidth), (
                    self.targetHeight * self.scaleFullyRate / newHeight))

            newWidth = math.ceil(newWidth * scaleRate)
            newHeight = math.ceil(newHeight * scaleRate)

        scaled = cv2.resize(image, (newHeight, newWidth), interpolation=self.rescalingInterpolation)

        # 将缩放后的符号居中放置在目标尺寸画布上
        x_add, y_add = (self.targetWidth - newWidth) // 2, (self.targetHeight - newHeight) // 2
        resized = np.zeros((self.targetWidth, self.targetHeight)) if len(image.shape) == 2 else np.zeros((self.targetWidth, self.targetHeight, 3))
        for x in range(newWidth):
            for y in range(newHeight):
                resized[x + x_add][y + y_add] = scaled[x][y]

        return resized

    def parseAndConvert(self, image_name: str) -> list:
        imagesInDotes = self.parseImage(image_name)
        original = 255 - cv2.imread(image_name)

        images = []

        for dotes in imagesInDotes:
            image = self._getImageAndCenterFromDotes(dotes, original)
            images.append([self.scaleParsedImage(image[0]), image[1]])
        rawImages = []
        for image, center in images:
            rawImages.append(RawImage(image, center))

        return rawImages

内容的提问来源于stack exchange,提问作者Andrew

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.12 20:01:06