You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于R语言tesseract包的相机陷阱图片OCR识别优化问询

改进R语言tesseract包识别相机陷阱图片信息栏的准确率方案

我有一批无元数据的相机陷阱图片,使用R语言的tesseract包OCR识别图片信息栏中的时间字符(数字、冒号、AM/PM),但识别错误率极高。已遵循OCR图片质量优化建议,调整过tesseract的页面分割模式、字符白名单等参数,仍无明显改善。曾尝试训练tesseract,但无法正确使用jTessBoxEditor工具。

原始代码如下:

########## Read text from JPEG images and rename files
# Last modified by JYL on 13 July 2022
# Description: This code loads JPEG images from camera traps, crops them to just the info bar
# and reads the text from that cropped image into a dataframe. 

# clear the environment & set the seed
remove(list = ls())
set.seed(2583722)

# Load libraries
library(tesseract)
library(magick)
library(dplyr)
library(tidytext)
library(tidyverse)
library(lubridate)

library(camtrapR)
library(exiftoolr)

# get set up
getwd()
wd_raw_images <- getwd()
camera <- 06
date = "06/18/2022"

# Read in the images
allFiles = list.files(path = getwd(), 
                      pattern = ".jpg", full.names = TRUE,
                      recursive = FALSE)

gc(full = TRUE)

# Get the dataframe with info
start_time <- Sys.time()
input <- magick::image_read(allFiles)
allInfo = image_info(input)
end_time <- Sys.time()
end_time - start_time

head(allInfo)
gc(full = TRUE)

# Attach the file names
allInfo$fileName = list.files(path = getwd(), pattern = ".jpg")
head(allInfo)

# Add row number as a column
allInfo <- allInfo %>% 
  mutate(row_id=row_number())
head(allInfo)

# Crop all images in the folder
start_time <- Sys.time()
text2 <- input %>%
  image_crop("328x49+2332+1484") %>%
  image_quantize(colorspace = "gray") %>%
  image_negate()
end_time <- Sys.time()
end_time - start_time 
text2[1]

gc(full = TRUE)

text2.2 <- text2 %>%
  image_scale("1640x245")
text2.2[1]

start_time <- Sys.time()
text2_2 <- text2.2 %>%
  image_transparent(color = "white", fuzz = 60) %>% 
  image_background("white") %>%
  image_threshold(type = "white", threshold = "50%") %>%
  image_threshold(type = "black", threshold = "50%")
end_time <- Sys.time()
end_time - start_time 
text2_2[1]

gc(full = TRUE)

# Limit what characters tesseract sees
params <- tesseract_params("pageseg")

numbers <- tesseract(options = list(tessedit_pageseg_mode = 13,
                                    tessedit_char_whitelist = ":0123456789APM",
                                    load_system_dawg = 0,
                                    load_freq_dawg = 0,
                                    textord_space_size_is_variable = 0),
                     cache = TRUE)

start_time <- Sys.time()
text3 <- tesseract::ocr(text2_2, engine = numbers)
end_time <- Sys.time()
end_time - start_time 
text3[1:10]

# Turn tesseract output into a dataframe
text_1_df <- data.frame(text = read.delim(textConnection(text3),
                                          header = FALSE, 
                                          sep = "", 
                                          strip.white = TRUE))

text_1_df <- text_1_df %>% 
  mutate(row_id=row_number()) %>%
  rename(Time = V1) %>%
  add_column(Date = date)%>%
  add_column(Camera = camera) 
head(text_1_df)

text_1_df$Time <- gsub(".*2022","",as.character(text_1_df$Time))
head(text_1_df)

datetime <- tibble(text_1_df)
head(datetime)

# save the dataframe
write.csv(datetime, "UncleanImageData_JPG.csv", row.names = TRUE, quote = FALSE)

改进方案

一、优化图片预处理流程

相机陷阱图片常存在局部明暗不均、噪点问题,调整预处理步骤可提升识别基础:

# 优化后的预处理步骤
text2_optimized <- input %>%
  image_crop("328x49+2332+1484") %>% # 保留原有裁剪区域
  image_quantize(colorspace = "gray") %>%
  image_negate() %>%
  image_scale("2000x300") %>% # 适度放大,保证字符边缘清晰
  image_blur(radius = 1, sigma = 0.5) %>% # 轻微模糊去除高频噪点
  image_adaptive_threshold(width = 15, height = 15, offset = 5) %>% # 自适应阈值处理局部明暗
  image_trim() %>% # 裁剪边缘空白区域
  image_threshold(type = "black", threshold = "60%") # 最后强化黑白对比

二、精细化调整Tesseract参数

针对固定格式的时间文本,调整参数强化数字识别优先级:

# 优化后的tesseract引擎配置
numbers_optimized <- tesseract(options = list(
  tessedit_pageseg_mode = 7, # 强制识别单行文本块,适合固定格式信息栏
  tessedit_char_whitelist = ":0123456789APM",
  load_system_dawg = 0,
  load_freq_dawg = 0,
  load_punc_dawg = 0,
  load_word_dawg = 0,
  classify_bln_numeric_mode = 1, # 优先识别数字模式
  textord_space_size_is_variable = 0,
  min_char_whitelist_conf = 0.8 # 过滤置信度低于80%的字符
), cache = TRUE)

# 执行OCR
text3_optimized <- tesseract::ocr(text2_optimized, engine = numbers_optimized)

三、后处理修正识别错误

利用时间格式的规则性,通过正则和字符替换修正常见识别错误:

# 时间文本清洗函数
clean_time <- function(raw_text) {
  # 替换常见识别错误字符
  cleaned <- raw_text %>%
    gsub("[O]", "0", .) %>% # 把O替换为0
    gsub("[lI]", "1", .) %>% # 把l/I替换为1
    gsub("[N]", "M", .) %>% # 把N替换为M
    gsub("[^:0-9APM ]", "", .) # 过滤白名单外的字符
  
  # 提取符合时间格式的内容
  time_match <- stringr::str_extract(cleaned, "\\d{1,2}:\\d{2}:\\d{2} [AP]M")
  
  # 标记无效结果,方便后续手动修正
  ifelse(is.na(time_match), paste0("INVALID: ", cleaned), time_match)
}

# 应用清洗函数到数据框
text_1_df$Cleaned_Time <- sapply(text_1_df$Time, clean_time)

四、替代训练方案(无需jTessBoxEditor)

若仍需自定义模型,可使用tesseract内置训练工具简化流程:

  1. 收集20-30张清晰的裁剪后信息栏样本
  2. 手动标注每个样本的正确时间文本
  3. 使用tesseract_train()函数训练自定义模型(需提前安装Tesseract训练工具包)

若觉得训练成本高,可尝试调用Python的easyocr(R中通过reticulate包集成),该工具对固定格式的数字文本识别表现更稳定。


内容的提问来源于stack exchange,提问作者Jen Lamb

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.28 14:17:50