You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

求助:如何用Python批量将PNG转文本并生成可搜索PDF/Word?

Hey there! Let's tackle your batch OCR problem step by step. First, I noticed a few typos and inconsistencies in your original code—let's fix those first, then add the batch processing features you need for CSV/Word output or a searchable PDF.

First: Fixing Your Original PDF-to-PNG Code

Here's the cleaned-up version of your initial code (note the corrected imports, variable names, and filename handling to avoid overwriting images):

import os
import sys
from PIL import Image
import pytesseract
from pdf2image import convert_from_path

# Update these paths with your actual directories
lib_path = r'_______'  # site-packages path (if needed)
poppler_path = r'_______'  # poppler dlls directory
pdf_path = r'_______'  # Input scanned PDF path
img_path = r'_______'  # Output folder for PNGs

# Add site-packages to path if necessary
sys.path.insert(0, lib_path)

# Convert PDF to PNGs (each page as separate file)
images = convert_from_path(pdf_path=pdf_path, dpi=500, poppler_path=poppler_path)

for idx, img in enumerate(images):
    # Save each page with unique filename (e.g., PDF_Page_1.png)
    img.save(os.path.join(img_path, f'PDF_Page_{idx+1}.png'), "PNG")
    print(f'Page {idx+1} converted to PNG')

Solution 1: Batch Convert PNGs to CSV

This script will loop through all your PNGs, extract text, and save everything to a CSV with page numbers and corresponding text:

import os
import pytesseract
from PIL import Image
import csv

# Configure paths
img_dir = r'_______'  # Folder with your PNGs
output_csv = r'_______/ocr_results.csv'

# Sort PNG files by page number to maintain order
png_files = sorted(
    [f for f in os.listdir(img_dir) if f.endswith('.png') and 'PDF_Page_' in f],
    key=lambda x: int(x.split('_')[2].split('.')[0])
)

# Write to CSV
with open(output_csv, 'w', newline='', encoding='utf-8') as csv_file:
    writer = csv.writer(csv_file)
    writer.writerow(['Page Number', 'Extracted Text'])  # Header row

    for file in png_files:
        full_img_path = os.path.join(img_dir, file)
        page_num = file.split('_')[2].split('.')[0]
        
        try:
            # Add lang='chi_sim' if you're processing Chinese text; remove for English
            text = pytesseract.image_to_string(Image.open(full_img_path), lang='chi_sim')
            writer.writerow([page_num, text.strip()])
            print(f'Processed page {page_num}')
        except Exception as e:
            print(f'Failed to process page {page_num}: {str(e)}')

Solution 2: Batch Convert PNGs to Word Document

Use python-docx to create a formatted Word doc with each page's text (install first with pip install python-docx):

import os
import pytesseract
from PIL import Image
from docx import Document

# Configure paths
img_dir = r'_______'
output_docx = r'_______/ocr_results.docx'

# Initialize Word document
doc = Document()
doc.add_heading('Scanned PDF OCR Results', level=1)

# Sort PNG files
png_files = sorted(
    [f for f in os.listdir(img_dir) if f.endswith('.png') and 'PDF_Page_' in f],
    key=lambda x: int(x.split('_')[2].split('.')[0])
)

for file in png_files:
    full_img_path = os.path.join(img_dir, file)
    page_num = file.split('_')[2].split('.')[0]
    
    try:
        text = pytesseract.image_to_string(Image.open(full_img_path), lang='chi_sim')
        # Add page heading and text
        doc.add_heading(f'Page {page_num}', level=2)
        doc.add_paragraph(text.strip())
        doc.add_page_break()  # Add page break after each page
        print(f'Processed page {page_num}')
    except Exception as e:
        print(f'Failed to process page {page_num}: {str(e)}')

# Save the Word document
doc.save(output_docx)

Solution 3: Generate a Searchable PDF

Use PyMuPDF (fitz) to overlay the extracted text onto your original scanned PDF, making it searchable (install first with pip install pymupdf):

import os
import pytesseract
from PIL import Image
import fitz  # PyMuPDF

# Configure paths
original_pdf = r'_______'  # Your original scanned PDF
img_dir = r'_______'  # Folder with PNGs
output_searchable_pdf = r'_______/searchable_pdf.pdf'

# Open original PDF
doc = fitz.open(original_pdf)

# Sort PNG files
png_files = sorted(
    [f for f in os.listdir(img_dir) if f.endswith('.png') and 'PDF_Page_' in f],
    key=lambda x: int(x.split('_')[2].split('.')[0])
)

for idx, file in enumerate(png_files):
    if idx >= len(doc):
        break  # Stop if we have more images than PDF pages
    
    full_img_path = os.path.join(img_dir, file)
    page = doc[idx]
    
    try:
        text = pytesseract.image_to_string(Image.open(full_img_path), lang='chi_sim')
        # Insert transparent text box over the entire page (so text is searchable but invisible)
        page_rect = fitz.Rect(0, 0, page.rect.width, page.rect.height)
        page.insert_textbox(page_rect, text, fontsize=10, color=(0,0,0), overlay=False)
        print(f'Processed page {idx+1}')
    except Exception as e:
        print(f'Failed to process page {idx+1}: {str(e)}')

# Save the searchable PDF
doc.save(output_searchable_pdf)
doc.close()

Quick Notes for Success:

  • Tesseract Path: If you're on Windows, add this line to point to your Tesseract executable:
    pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'
    
  • Language Support: For non-English text (like Chinese), install the corresponding Tesseract language pack and use the lang parameter in image_to_string.
  • Dependencies: Install all required packages with:
    pip install pytesseract pillow pdf2image python-docx pymupdf
    

内容的提问来源于stack exchange,提问作者Deepak

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.14 08:33:08