求助:如何用Python批量将PNG转文本并生成可搜索PDF/Word?
Hey there! Let's tackle your batch OCR problem step by step. First, I noticed a few typos and inconsistencies in your original code—let's fix those first, then add the batch processing features you need for CSV/Word output or a searchable PDF.
First: Fixing Your Original PDF-to-PNG Code
Here's the cleaned-up version of your initial code (note the corrected imports, variable names, and filename handling to avoid overwriting images):
import os import sys from PIL import Image import pytesseract from pdf2image import convert_from_path # Update these paths with your actual directories lib_path = r'_______' # site-packages path (if needed) poppler_path = r'_______' # poppler dlls directory pdf_path = r'_______' # Input scanned PDF path img_path = r'_______' # Output folder for PNGs # Add site-packages to path if necessary sys.path.insert(0, lib_path) # Convert PDF to PNGs (each page as separate file) images = convert_from_path(pdf_path=pdf_path, dpi=500, poppler_path=poppler_path) for idx, img in enumerate(images): # Save each page with unique filename (e.g., PDF_Page_1.png) img.save(os.path.join(img_path, f'PDF_Page_{idx+1}.png'), "PNG") print(f'Page {idx+1} converted to PNG')
Solution 1: Batch Convert PNGs to CSV
This script will loop through all your PNGs, extract text, and save everything to a CSV with page numbers and corresponding text:
import os import pytesseract from PIL import Image import csv # Configure paths img_dir = r'_______' # Folder with your PNGs output_csv = r'_______/ocr_results.csv' # Sort PNG files by page number to maintain order png_files = sorted( [f for f in os.listdir(img_dir) if f.endswith('.png') and 'PDF_Page_' in f], key=lambda x: int(x.split('_')[2].split('.')[0]) ) # Write to CSV with open(output_csv, 'w', newline='', encoding='utf-8') as csv_file: writer = csv.writer(csv_file) writer.writerow(['Page Number', 'Extracted Text']) # Header row for file in png_files: full_img_path = os.path.join(img_dir, file) page_num = file.split('_')[2].split('.')[0] try: # Add lang='chi_sim' if you're processing Chinese text; remove for English text = pytesseract.image_to_string(Image.open(full_img_path), lang='chi_sim') writer.writerow([page_num, text.strip()]) print(f'Processed page {page_num}') except Exception as e: print(f'Failed to process page {page_num}: {str(e)}')
Solution 2: Batch Convert PNGs to Word Document
Use python-docx to create a formatted Word doc with each page's text (install first with pip install python-docx):
import os import pytesseract from PIL import Image from docx import Document # Configure paths img_dir = r'_______' output_docx = r'_______/ocr_results.docx' # Initialize Word document doc = Document() doc.add_heading('Scanned PDF OCR Results', level=1) # Sort PNG files png_files = sorted( [f for f in os.listdir(img_dir) if f.endswith('.png') and 'PDF_Page_' in f], key=lambda x: int(x.split('_')[2].split('.')[0]) ) for file in png_files: full_img_path = os.path.join(img_dir, file) page_num = file.split('_')[2].split('.')[0] try: text = pytesseract.image_to_string(Image.open(full_img_path), lang='chi_sim') # Add page heading and text doc.add_heading(f'Page {page_num}', level=2) doc.add_paragraph(text.strip()) doc.add_page_break() # Add page break after each page print(f'Processed page {page_num}') except Exception as e: print(f'Failed to process page {page_num}: {str(e)}') # Save the Word document doc.save(output_docx)
Solution 3: Generate a Searchable PDF
Use PyMuPDF (fitz) to overlay the extracted text onto your original scanned PDF, making it searchable (install first with pip install pymupdf):
import os import pytesseract from PIL import Image import fitz # PyMuPDF # Configure paths original_pdf = r'_______' # Your original scanned PDF img_dir = r'_______' # Folder with PNGs output_searchable_pdf = r'_______/searchable_pdf.pdf' # Open original PDF doc = fitz.open(original_pdf) # Sort PNG files png_files = sorted( [f for f in os.listdir(img_dir) if f.endswith('.png') and 'PDF_Page_' in f], key=lambda x: int(x.split('_')[2].split('.')[0]) ) for idx, file in enumerate(png_files): if idx >= len(doc): break # Stop if we have more images than PDF pages full_img_path = os.path.join(img_dir, file) page = doc[idx] try: text = pytesseract.image_to_string(Image.open(full_img_path), lang='chi_sim') # Insert transparent text box over the entire page (so text is searchable but invisible) page_rect = fitz.Rect(0, 0, page.rect.width, page.rect.height) page.insert_textbox(page_rect, text, fontsize=10, color=(0,0,0), overlay=False) print(f'Processed page {idx+1}') except Exception as e: print(f'Failed to process page {idx+1}: {str(e)}') # Save the searchable PDF doc.save(output_searchable_pdf) doc.close()
Quick Notes for Success:
- Tesseract Path: If you're on Windows, add this line to point to your Tesseract executable:
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe' - Language Support: For non-English text (like Chinese), install the corresponding Tesseract language pack and use the
langparameter inimage_to_string. - Dependencies: Install all required packages with:
pip install pytesseract pillow pdf2image python-docx pymupdf
内容的提问来源于stack exchange,提问作者Deepak

