You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

从指定参数图片提取URL与错误信息的技术实现问询

Extract URLs and Error Messages from 300 DPI 1536×1068 Images

Here's a polished, efficient version of your code that meets your requirements. I've fixed common pitfalls (like redundant imports) and added validation, preprocessing, and regex extraction logic tailored to your use case:

import glob
import os
import re
import cv2
import pytesseract
from PIL import Image
from matplotlib import pyplot as plt

# Configuration constants
INPUT_DIR = '/home/agnihotri/img/*.jpg'
PROCESSED_DIR = 'processed/images'
TARGET_WIDTH = 1536
TARGET_HEIGHT = 1068
TARGET_DPI = 300

# Create processed directory if it doesn't exist
if not os.path.exists(PROCESSED_DIR):
    os.makedirs(PROCESSED_DIR)

# Regex patterns for extracting target content
URL_PATTERN = r'https?://\S+|www\.\S+'
ERROR_PATTERN = r'(Error|Failed|Exception|Error:)\s*.+'

def validate_image(image_path):
    """Check if image matches target dimensions and DPI requirements"""
    with Image.open(image_path) as img:
        # Check dimensions (PIL uses (width, height) order)
        if img.size != (TARGET_WIDTH, TARGET_HEIGHT):
            return False, f"Invalid dimensions: {img.size} vs expected ({TARGET_WIDTH}, {TARGET_HEIGHT})"
        # Check DPI from EXIF metadata
        dpi = img.info.get('dpi', (0, 0))
        if dpi[0] != TARGET_DPI or dpi[1] != TARGET_DPI:
            return False, f"Invalid DPI: {dpi} vs expected ({TARGET_DPI}, {TARGET_DPI})"
    return True, "Valid image"

def preprocess_image(image):
    """Preprocess image to improve OCR accuracy"""
    # Convert to grayscale to simplify text detection
    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
    # Apply Gaussian blur to reduce noise
    blurred = cv2.GaussianBlur(gray, (3, 3), 0)
    # Adaptive thresholding to create high-contrast text
    thresh = cv2.adaptiveThreshold(blurred, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY_INV, 11, 2)
    return thresh

def extract_info(text):
    """Extract URLs and error messages from OCR-generated text"""
    urls = re.findall(URL_PATTERN, text)
    # Capture error lines (case-insensitive to catch variations like "error" or "ERROR")
    errors = re.findall(ERROR_PATTERN, text, re.IGNORECASE)
    # Clean up tuple matches from regex group captures
    errors = [err[0] + err[1] if isinstance(err, tuple) else err for err in errors]
    return urls, errors

# Main execution flow
list_f = glob.glob(INPUT_DIR)
res_final = []

if not list_f:
    print("No JPG files found in the input directory.")
else:
    for f in list_f:
        filename = os.path.basename(f)
        print(f"Processing {filename}...")
        
        # Skip invalid images upfront
        is_valid, msg = validate_image(f)
        if not is_valid:
            print(f"Skipping {filename}: {msg}")
            res_final.append({
                'filename': filename,
                'valid': False,
                'validation_error': msg,
                'urls': [],
                'errors': [],
                'full_text': ''
            })
            continue
        
        # Load and preprocess image for OCR
        image = cv2.imread(f)
        processed_img = preprocess_image(image)
        
        # Save processed image for reference
        processed_path = os.path.join(PROCESSED_DIR, filename)
        cv2.imwrite(processed_path, processed_img)
        
        # Extract text with Tesseract (config assumes uniform text block)
        custom_config = r'--oem 3 --psm 6'
        full_text = pytesseract.image_to_string(processed_img, config=custom_config)
        
        # Extract target content
        urls, errors = extract_info(full_text)
        
        # Store structured results
        res_final.append({
            'filename': filename,
            'valid': True,
            'urls': urls,
            'errors': errors,
            'full_text': full_text.strip()
        })
    
    # Print summary of results
    print("\nProcessing complete! Summary:")
    for result in res_final:
        print(f"\nFile: {result['filename']}")
        if result['valid']:
            print(f"URLs found: {len(result['urls'])}")
            if result['urls']:
                for url in result['urls']:
                    print(f"- {url}")
            print(f"Errors found: {len(result['errors'])}")
            if result['errors']:
                for error in result['errors']:
                    print(f"- {error}")
        else:
            print(f"Validation failed: {result['validation_error']}")

Key Improvements & Explanations:

  • Efficient imports: Moved all imports to the top to avoid redundant work in each loop iteration.
  • Strict validation: Checks both image dimensions and DPI using PIL's EXIF reading to ensure only your target images are processed.
  • OCR preprocessing: Converts images to grayscale, reduces noise, and applies thresholding to create high-contrast text that Tesseract can read more reliably.
  • Custom Tesseract config: Uses --psm 6 to assume a single uniform text block (adjust the psm value if your images have multiple disjoint text regions).
  • Targeted regex extraction:
    • URL pattern catches both http(s):// and www. formatted URLs.
    • Error pattern looks for common error keywords (case-insensitive) and captures full error lines.
  • Structured results: Stores all data in a list of dictionaries for easy post-processing (e.g., saving to CSV/JSON).
  • Progress feedback: Prints real-time processing status and a clear summary at the end.

Dependencies Note:

Make sure you have these packages installed:

pip install pillow opencv-python pytesseract matplotlib

Also, ensure the Tesseract OCR engine is installed on your system (e.g., sudo apt install tesseract-ocr on Ubuntu, or download from the official Tesseract repository for Windows/macOS).

内容的提问来源于stack exchange,提问作者rahulagnihotri

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.25 07:35:19