基于R语言tesseract包的相机陷阱图片OCR识别优化问询
改进R语言tesseract包识别相机陷阱图片信息栏的准确率方案
我有一批无元数据的相机陷阱图片,使用R语言的tesseract包OCR识别图片信息栏中的时间字符(数字、冒号、AM/PM),但识别错误率极高。已遵循OCR图片质量优化建议,调整过tesseract的页面分割模式、字符白名单等参数,仍无明显改善。曾尝试训练tesseract,但无法正确使用jTessBoxEditor工具。
原始代码如下:
########## Read text from JPEG images and rename files # Last modified by JYL on 13 July 2022 # Description: This code loads JPEG images from camera traps, crops them to just the info bar # and reads the text from that cropped image into a dataframe. # clear the environment & set the seed remove(list = ls()) set.seed(2583722) # Load libraries library(tesseract) library(magick) library(dplyr) library(tidytext) library(tidyverse) library(lubridate) library(camtrapR) library(exiftoolr) # get set up getwd() wd_raw_images <- getwd() camera <- 06 date = "06/18/2022" # Read in the images allFiles = list.files(path = getwd(), pattern = ".jpg", full.names = TRUE, recursive = FALSE) gc(full = TRUE) # Get the dataframe with info start_time <- Sys.time() input <- magick::image_read(allFiles) allInfo = image_info(input) end_time <- Sys.time() end_time - start_time head(allInfo) gc(full = TRUE) # Attach the file names allInfo$fileName = list.files(path = getwd(), pattern = ".jpg") head(allInfo) # Add row number as a column allInfo <- allInfo %>% mutate(row_id=row_number()) head(allInfo) # Crop all images in the folder start_time <- Sys.time() text2 <- input %>% image_crop("328x49+2332+1484") %>% image_quantize(colorspace = "gray") %>% image_negate() end_time <- Sys.time() end_time - start_time text2[1] gc(full = TRUE) text2.2 <- text2 %>% image_scale("1640x245") text2.2[1] start_time <- Sys.time() text2_2 <- text2.2 %>% image_transparent(color = "white", fuzz = 60) %>% image_background("white") %>% image_threshold(type = "white", threshold = "50%") %>% image_threshold(type = "black", threshold = "50%") end_time <- Sys.time() end_time - start_time text2_2[1] gc(full = TRUE) # Limit what characters tesseract sees params <- tesseract_params("pageseg") numbers <- tesseract(options = list(tessedit_pageseg_mode = 13, tessedit_char_whitelist = ":0123456789APM", load_system_dawg = 0, load_freq_dawg = 0, textord_space_size_is_variable = 0), cache = TRUE) start_time <- Sys.time() text3 <- tesseract::ocr(text2_2, engine = numbers) end_time <- Sys.time() end_time - start_time text3[1:10] # Turn tesseract output into a dataframe text_1_df <- data.frame(text = read.delim(textConnection(text3), header = FALSE, sep = "", strip.white = TRUE)) text_1_df <- text_1_df %>% mutate(row_id=row_number()) %>% rename(Time = V1) %>% add_column(Date = date)%>% add_column(Camera = camera) head(text_1_df) text_1_df$Time <- gsub(".*2022","",as.character(text_1_df$Time)) head(text_1_df) datetime <- tibble(text_1_df) head(datetime) # save the dataframe write.csv(datetime, "UncleanImageData_JPG.csv", row.names = TRUE, quote = FALSE)
改进方案
一、优化图片预处理流程
相机陷阱图片常存在局部明暗不均、噪点问题,调整预处理步骤可提升识别基础:
# 优化后的预处理步骤 text2_optimized <- input %>% image_crop("328x49+2332+1484") %>% # 保留原有裁剪区域 image_quantize(colorspace = "gray") %>% image_negate() %>% image_scale("2000x300") %>% # 适度放大,保证字符边缘清晰 image_blur(radius = 1, sigma = 0.5) %>% # 轻微模糊去除高频噪点 image_adaptive_threshold(width = 15, height = 15, offset = 5) %>% # 自适应阈值处理局部明暗 image_trim() %>% # 裁剪边缘空白区域 image_threshold(type = "black", threshold = "60%") # 最后强化黑白对比
二、精细化调整Tesseract参数
针对固定格式的时间文本,调整参数强化数字识别优先级:
# 优化后的tesseract引擎配置 numbers_optimized <- tesseract(options = list( tessedit_pageseg_mode = 7, # 强制识别单行文本块,适合固定格式信息栏 tessedit_char_whitelist = ":0123456789APM", load_system_dawg = 0, load_freq_dawg = 0, load_punc_dawg = 0, load_word_dawg = 0, classify_bln_numeric_mode = 1, # 优先识别数字模式 textord_space_size_is_variable = 0, min_char_whitelist_conf = 0.8 # 过滤置信度低于80%的字符 ), cache = TRUE) # 执行OCR text3_optimized <- tesseract::ocr(text2_optimized, engine = numbers_optimized)
三、后处理修正识别错误
利用时间格式的规则性,通过正则和字符替换修正常见识别错误:
# 时间文本清洗函数 clean_time <- function(raw_text) { # 替换常见识别错误字符 cleaned <- raw_text %>% gsub("[O]", "0", .) %>% # 把O替换为0 gsub("[lI]", "1", .) %>% # 把l/I替换为1 gsub("[N]", "M", .) %>% # 把N替换为M gsub("[^:0-9APM ]", "", .) # 过滤白名单外的字符 # 提取符合时间格式的内容 time_match <- stringr::str_extract(cleaned, "\\d{1,2}:\\d{2}:\\d{2} [AP]M") # 标记无效结果,方便后续手动修正 ifelse(is.na(time_match), paste0("INVALID: ", cleaned), time_match) } # 应用清洗函数到数据框 text_1_df$Cleaned_Time <- sapply(text_1_df$Time, clean_time)
四、替代训练方案(无需jTessBoxEditor)
若仍需自定义模型,可使用tesseract内置训练工具简化流程:
- 收集20-30张清晰的裁剪后信息栏样本
- 手动标注每个样本的正确时间文本
- 使用
tesseract_train()函数训练自定义模型(需提前安装Tesseract训练工具包)
若觉得训练成本高,可尝试调用Python的easyocr(R中通过reticulate包集成),该工具对固定格式的数字文本识别表现更稳定。
内容的提问来源于stack exchange,提问作者Jen Lamb
相关产品推荐
相关产品推荐

