Python批量提取图片文本存CSV仅单条记录的问题解决
批量提取图片文本后CSV仅保留最后一条记录的解决方法
需求背景
- 使用Python提取文件夹中所有图片的文本并存储
- 后续需将提取的文本分类为垃圾/非垃圾信息
原始代码
from PIL import Image import pytesseract as pt import pandas as pd from tabulate import tabulate from io import StringIO import os import json import csv def main(): # path for the folder for getting the raw images path = "E:/mehr mtech p1/images/" # link to the file in which output needs to be kept fullTempPath = "E:/mehr mtech p1/out.txt" # iterating the images inside the folder for imageName in os.listdir(path): inputPath = os.path.join(path, imageName) img = Image.open(inputPath) #print(imageName) # applying ocr using pytesseract for python pt.pytesseract.tesseract_cmd = r'C:/Program Files/Tesseract-OCR/tesseract.exe' text = pt.image_to_string(img, lang = "eng") #print(text) dictionary = {'image': imageName, 'Text': text} print(dictionary) #Create a datafrmae from the dictionary df = pd.DataFrame(dictionary, index=[0]) #print dataframe. #print(df) #print(tabulate(df, headers = 'keys', tablefmt = 'psql')) #Creating a string of the dictionary to print the data with labels in string format in the txt file #string = json.dumps(dictionary) #f1 = open("E:/mehr mtech p1/mmyfile.txt","a+") #f1.write(string) #df = pd.read_csv(string, sep = ";") #print(df) df.to_csv("E:/mehr mtech p1/tableimage.csv") # saving the text for appending it to the output.txt file # a + parameter used for creating the file if not present # and if present then append the text content file1 = open(fullTempPath, "a+") # providing the name of the image file1.write(imageName+"\n") # providing the content in the image file1.write(text+"\n") file1.close() # for printing the output file file2 = open(fullTempPath, 'r') print(file2.read()) file2.close() if __name__ == '__main__': main()
问题描述
运行上述代码批量提取图片文本时,每张图片的文件名和提取文本都能正确存入字典,但保存为CSV文件时,最终文件里仅保留最后一张图片的记录。
问题原因
每次循环中都会创建一个仅包含当前图片数据的新DataFrame,调用df.to_csv()时默认会覆盖已存在的CSV文件,循环结束后自然只剩下最后一次写入的记录。
解决方法
方法1:先收集所有数据再统一保存
先初始化空列表,循环中将每个图片的字典数据添加到列表,循环结束后再将整个列表转为DataFrame并保存到CSV,一次性写入所有记录。
修改后的完整代码:
from PIL import Image import pytesseract as pt import pandas as pd import os def main(): path = "E:/mehr mtech p1/images/" fullTempPath = "E:/mehr mtech p1/out.txt" # 初始化空列表存储所有图片数据 data_list = [] for imageName in os.listdir(path): inputPath = os.path.join(path, imageName) img = Image.open(inputPath) pt.pytesseract.tesseract_cmd = r'C:/Program Files/Tesseract-OCR/tesseract.exe' text = pt.image_to_string(img, lang = "eng") dictionary = {'image': imageName, 'Text': text} print(dictionary) # 将当前数据加入列表 data_list.append(dictionary) # 写入txt文件逻辑不变 file1 = open(fullTempPath, "a+") file1.write(imageName+"\n") file1.write(text+"\n") file1.close() # 循环结束后统一生成DataFrame并保存CSV df = pd.DataFrame(data_list) df.to_csv("E:/mehr mtech p1/tableimage.csv", index=False) # 打印txt内容逻辑不变 file2 = open(fullTempPath, 'r') print(file2.read()) file2.close() if __name__ == '__main__': main()
方法2:循环中追加写入CSV
如果不想一次性收集所有数据,可在每次调用to_csv()时设置追加模式,控制表头仅在第一次写入时添加。
修改后的关键循环片段:
def main(): path = "E:/mehr mtech p1/images/" fullTempPath = "E:/mehr mtech p1/out.txt" csv_path = "E:/mehr mtech p1/tableimage.csv" # 检查CSV是否已存在,用于判断是否写入表头 file_exists = os.path.isfile(csv_path) for imageName in os.listdir(path): inputPath = os.path.join(path, imageName) img = Image.open(inputPath) pt.pytesseract.tesseract_cmd = r'C:/Program Files/Tesseract-OCR/tesseract.exe' text = pt.image_to_string(img, lang = "eng") dictionary = {'image': imageName, 'Text': text} print(dictionary) df = pd.DataFrame(dictionary, index=[0]) # 追加模式写入,第一次写入表头,后续跳过 df.to_csv(csv_path, mode='a', header=not file_exists, index=False) # 第一次写入后标记文件已存在 if not file_exists: file_exists = True # 写入txt文件逻辑不变 file1 = open(fullTempPath, "a+") file1.write(imageName+"\n") file1.write(text+"\n") file1.close() # 打印txt内容逻辑不变 file2 = open(fullTempPath, 'r') print(file2.read()) file2.close()
说明
- 方法1更高效,减少文件IO操作,适合处理大量图片
- 方法2适合内存有限的场景,逐行追加写入数据
内容的提问来源于stack exchange,提问作者Paridhi Kaushik
相关产品推荐
相关产品推荐

