从指定参数图片提取URL与错误信息的技术实现问询
Extract URLs and Error Messages from 300 DPI 1536×1068 Images
Here's a polished, efficient version of your code that meets your requirements. I've fixed common pitfalls (like redundant imports) and added validation, preprocessing, and regex extraction logic tailored to your use case:
import glob import os import re import cv2 import pytesseract from PIL import Image from matplotlib import pyplot as plt # Configuration constants INPUT_DIR = '/home/agnihotri/img/*.jpg' PROCESSED_DIR = 'processed/images' TARGET_WIDTH = 1536 TARGET_HEIGHT = 1068 TARGET_DPI = 300 # Create processed directory if it doesn't exist if not os.path.exists(PROCESSED_DIR): os.makedirs(PROCESSED_DIR) # Regex patterns for extracting target content URL_PATTERN = r'https?://\S+|www\.\S+' ERROR_PATTERN = r'(Error|Failed|Exception|Error:)\s*.+' def validate_image(image_path): """Check if image matches target dimensions and DPI requirements""" with Image.open(image_path) as img: # Check dimensions (PIL uses (width, height) order) if img.size != (TARGET_WIDTH, TARGET_HEIGHT): return False, f"Invalid dimensions: {img.size} vs expected ({TARGET_WIDTH}, {TARGET_HEIGHT})" # Check DPI from EXIF metadata dpi = img.info.get('dpi', (0, 0)) if dpi[0] != TARGET_DPI or dpi[1] != TARGET_DPI: return False, f"Invalid DPI: {dpi} vs expected ({TARGET_DPI}, {TARGET_DPI})" return True, "Valid image" def preprocess_image(image): """Preprocess image to improve OCR accuracy""" # Convert to grayscale to simplify text detection gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) # Apply Gaussian blur to reduce noise blurred = cv2.GaussianBlur(gray, (3, 3), 0) # Adaptive thresholding to create high-contrast text thresh = cv2.adaptiveThreshold(blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY_INV, 11, 2) return thresh def extract_info(text): """Extract URLs and error messages from OCR-generated text""" urls = re.findall(URL_PATTERN, text) # Capture error lines (case-insensitive to catch variations like "error" or "ERROR") errors = re.findall(ERROR_PATTERN, text, re.IGNORECASE) # Clean up tuple matches from regex group captures errors = [err[0] + err[1] if isinstance(err, tuple) else err for err in errors] return urls, errors # Main execution flow list_f = glob.glob(INPUT_DIR) res_final = [] if not list_f: print("No JPG files found in the input directory.") else: for f in list_f: filename = os.path.basename(f) print(f"Processing {filename}...") # Skip invalid images upfront is_valid, msg = validate_image(f) if not is_valid: print(f"Skipping {filename}: {msg}") res_final.append({ 'filename': filename, 'valid': False, 'validation_error': msg, 'urls': [], 'errors': [], 'full_text': '' }) continue # Load and preprocess image for OCR image = cv2.imread(f) processed_img = preprocess_image(image) # Save processed image for reference processed_path = os.path.join(PROCESSED_DIR, filename) cv2.imwrite(processed_path, processed_img) # Extract text with Tesseract (config assumes uniform text block) custom_config = r'--oem 3 --psm 6' full_text = pytesseract.image_to_string(processed_img, config=custom_config) # Extract target content urls, errors = extract_info(full_text) # Store structured results res_final.append({ 'filename': filename, 'valid': True, 'urls': urls, 'errors': errors, 'full_text': full_text.strip() }) # Print summary of results print("\nProcessing complete! Summary:") for result in res_final: print(f"\nFile: {result['filename']}") if result['valid']: print(f"URLs found: {len(result['urls'])}") if result['urls']: for url in result['urls']: print(f"- {url}") print(f"Errors found: {len(result['errors'])}") if result['errors']: for error in result['errors']: print(f"- {error}") else: print(f"Validation failed: {result['validation_error']}")
Key Improvements & Explanations:
- Efficient imports: Moved all imports to the top to avoid redundant work in each loop iteration.
- Strict validation: Checks both image dimensions and DPI using PIL's EXIF reading to ensure only your target images are processed.
- OCR preprocessing: Converts images to grayscale, reduces noise, and applies thresholding to create high-contrast text that Tesseract can read more reliably.
- Custom Tesseract config: Uses
--psm 6to assume a single uniform text block (adjust thepsmvalue if your images have multiple disjoint text regions). - Targeted regex extraction:
- URL pattern catches both
http(s)://andwww.formatted URLs. - Error pattern looks for common error keywords (case-insensitive) and captures full error lines.
- URL pattern catches both
- Structured results: Stores all data in a list of dictionaries for easy post-processing (e.g., saving to CSV/JSON).
- Progress feedback: Prints real-time processing status and a clear summary at the end.
Dependencies Note:
Make sure you have these packages installed:
pip install pillow opencv-python pytesseract matplotlib
Also, ensure the Tesseract OCR engine is installed on your system (e.g., sudo apt install tesseract-ocr on Ubuntu, or download from the official Tesseract repository for Windows/macOS).
内容的提问来源于stack exchange,提问作者rahulagnihotri
相关产品推荐
相关产品推荐

