You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

R语言批量提取PDF中LOT编号失败:仅读取单个文件问题排查

批量提取PDF中LOT编号的代码问题排查

问题背景

尝试用R编写函数从数百份格式统一的PDF文档中提取LOT编号,单文件测试可正常返回结果,但封装为函数后批量运行失效,仅能处理单个PDF文件。以下为简化版代码及会话信息:

原代码

# Load the required library
library(tidyverse)
library(pdftools)

root_folder <- "P:\\Data Transfer\\Fortessa 245\\Lucas\\Achilles Base 2Blue 3Red 6Vio 3YG 4UV"

pdf_files <- list.files(path = root_folder, 
                        pattern = "*.pdf", 
                        recursive = TRUE, 
                        full.names = TRUE)

extract_cstLOT <- function(pdf_file){
  # read in a PDF file
  pdf_file <- pdf_text(pdf_file)
    # Use str_split to split the text into lines
  lines <- data.frame(str_split(pdf_file, "\n") %>% unlist())
    #extract CST lot from lines
  cstLOT <- lapply(lines$str_split.pdf_file....n.......unlist..[13], function(x) {
    split_result <- str_split(x, "  ")
    if (length(split_result) == 0) {
      return(list(x))
    } else {
      return(split_result[[1]])
    }
  })
  # Remove empty elements from each column
  cstLOT <- lapply(cstLOT, function(x) x[x != ""])
  #convert to dataframe
  cstLOT <- t(data.frame(cstLOT))
  #remove row names
  rownames(cstLOT)<-NULL
  #extract cst lot as a numeric object
  cstLOT <- as.numeric(cstLOT[,2])
  
  return(cstLOT)
}

all_CST<-as.data.frame("")
for (n in length(pdf_files)) {
  current_CST<-extract_cstLOT(pdf_file = pdf_files[n])
  all_CST<-cbind(all_CST, current_CST)
}

会话信息

R version 4.2.2 (2022-10-31 ucrt)
Platform: x86_64-w64-mingw32/x64 (64-bit)
Running under: Windows 10 x64 (build 19045)

Matrix products: default

locale:
[1] LC_COLLATE=English_United States.utf8  LC_CTYPE=English_United States.utf8    LC_MONETARY=English_United States.utf8 LC_NUMERIC=C                          
[5] LC_TIME=English_United States.utf8    

attached base packages:
[1] stats     graphics  grDevices utils     datasets  methods   base     

other attached packages:
 [1] pdftools_3.3.3  forcats_0.5.2   stringr_1.4.1   dplyr_1.0.10    purrr_0.3.5     readr_2.1.3     tidyr_1.2.1     tibble_3.1.8    ggplot2_3.4.0   tidyverse_1.3.2

loaded via a namespace (and not attached):
 [1] Rcpp_1.0.9          pillar_1.8.1        compiler_4.2.2      cellranger_1.1.0    dbplyr_2.2.1        tools_4.2.2         timechange_0.1.1    lubridate_1.9.0    
 [9] jsonlite_1.8.3      googledrive_2.0.0   lifecycle_1.0.3     gargle_1.2.1        gtable_0.3.1        pkgconfig_2.0.3     rlang_1.0.6         reprex_2.0.2       
[17] DBI_1.1.3           cli_3.4.1           rstudioapi_0.14     haven_2.5.1         xml2_1.3.3          withr_2.5.0         httr_1.4.4          askpass_1.1        
[25] generics_0.1.3      vctrs_0.5.1         fs_1.5.2            hms_1.1.2           googlesheets4_1.0.1 grid_4.2.2          tidyselect_1.2.0    glue_1.6.2         
[33] qpdf_1.3.0          R6_2.5.1            fansi_1.0.3         readxl_1.4.1        tzdb_0.3.0          modelr_0.1.10       magrittr_2.0.3      ellipsis_0.3.2     
[41] backports_1.4.1     scales_1.2.1        rvest_1.0.3         assertthat_0.2.1    colorspace_2.0-3    utf8_1.2.2          stringi_1.7.8       munsell_0.5.0      
[49] broom_1.0.1         crayon_1.5.2 

问题排查与修正

1. 循环逻辑错误

原循环for (n in length(pdf_files))仅会执行一次(因为length(pdf_files)返回单个数值,比如当有100个文件时,n只会等于100),无法遍历所有PDF文件。

修正:改为遍历文件索引序列:

for (n in seq_along(pdf_files)) {
  # ... 处理逻辑
}

2. 自动生成的列名导致索引失效

原代码中lines <- data.frame(str_split(pdf_file, "\n") %>% unlist())会生成一个列名为str_split.pdf_file....n.......unlist..的不规范列名,后续用lines$str_split.pdf_file....n.......unlist..[13]取第13行时,极易因列名识别错误导致失败。

修正:显式指定列名:

lines <- data.frame(text = str_split(pdf_file, "\n") %>% unlist())
target_line <- lines$text[13] # 用规范列名取目标行

3. 冗余的lapply嵌套

lines$text[13]是单个字符串,无需用lapply处理,直接对该字符串操作即可,冗余嵌套会增加逻辑复杂度。

修正:简化提取逻辑:

target_line <- lines$text[13]
split_result <- str_split(target_line, "  ")[[1]] %>% .[. != ""] # 拆分后移除空元素
cstLOT <- as.numeric(split_result[2])

4. 低效且易出错的数据拼接方式

原代码初始化all_CST<-as.data.frame("")并反复用cbind拼接,会引入空列且效率低下。

修正:改用rbind拼接带文件路径的结果(或用tidyverse的map_dfr批量处理):

# 用map_dfr批量处理(推荐)
all_CST <- map_dfr(pdf_files, function(file) {
  data.frame(file_path = file, cstLOT = extract_cstLOT(file))
})

# 或修正后的循环写法
all_CST <- data.frame(file_path = character(), cstLOT = numeric())
for (n in seq_along(pdf_files)) {
  current_file <- pdf_files[n]
  current_CST <- extract_cstLOT(current_file)
  all_CST <- rbind(all_CST, data.frame(file_path = current_file, cstLOT = current_CST))
}

完整修正后的代码

library(tidyverse)
library(pdftools)

root_folder <- "P:\\Data Transfer\\Fortessa 245\\Lucas\\Achilles Base 2Blue 3Red 6Vio 3YG 4UV"

pdf_files <- list.files(path = root_folder, 
                        pattern = "*.pdf", 
                        recursive = TRUE, 
                        full.names = TRUE)

extract_cstLOT <- function(pdf_file){
  # 读取PDF文本
  pdf_text_content <- pdf_text(pdf_file)
  # 按行拆分并转为带列名的数据框
  lines <- data.frame(text = str_split(pdf_text_content, "\n") %>% unlist())
  
  # 取第13行文本(需确认所有PDF格式统一,该行确实包含LOT信息)
  target_line <- lines$text[13]
  # 按双空格拆分并移除空元素
  split_result <- str_split(target_line, "  ")[[1]] %>% .[. != ""]
  
  # 提取第二部分转为数值
  cstLOT <- as.numeric(split_result[2])
  return(cstLOT)
}

# 批量处理所有PDF并生成结果数据框
all_CST <- map_dfr(pdf_files, function(file) {
  data.frame(file_path = file, cstLOT = extract_cstLOT(file))
})

内容的提问来源于stack exchange,提问作者ルーカス 黒

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.02 01:21:06