如何优化用于勾选/圈选类图像分类的CNN模型?
4类标记图像分类模型优化求助
我用CNN对4类图像做分类,类别分别是ticked(勾选)、unticked(未勾选)、circled_yes(圈选是)、circled_no(圈选否),目标是让模型精准判断图像类别。
数据集按类别分4个文件夹,通过调整分辨率(80x80、120x80)扩充数据,各类别数量相近,每类约340-360张。当前模型有2M参数,测试准确率能到98%-99%,但尝试减少模型参数、新增训练图像、图像增广这些方法时,模型性能反而下降,求有效的优化建议。
当前使用代码
import tensorflow as tf from tensorflow.keras import layers, models from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau from tensorflow.keras.optimizers import Adam import numpy as np import matplotlib.pyplot as plt import cv2 import os from sklearn.model_selection import train_test_split from sklearn.utils import class_weight from sklearn.metrics import classification_report import logging import random # Set the random seeds for reproducibility os.environ["PYTHONHASHSEED"] = "0" random.seed(42) np.random.seed(42) tf.random.set_seed(42) os.environ["TF_DETERMINISTIC_OPS"] = "1" # Initialize logger logging.basicConfig(level=logging.INFO) LOGGER = logging.getLogger(__name__) # Load and preprocess images def load_images_from_folder(folder): current_dir = os.getcwd() folder_path = os.path.join(current_dir, folder) LOGGER.info(f"Folder path is {folder_path}") images = [] labels = [] if not os.path.isdir(folder_path): raise ValueError(f"Folder {folder_path} does not exist.") for label in os.listdir(folder_path): LOGGER.info(f"Processing label: {label}") if label == "images_test": continue label_folder = os.path.join(folder_path, label) LOGGER.info(f"Label '{label}' folder is {label_folder}") if not os.path.isdir(label_folder): continue # Walk through the label folder and its subfolders for root, dirs, files in os.walk(label_folder): for file in files: img_path = os.path.join(root, file) # LOGGER.info(f"Processing image {img_path}") img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE) # Load image in grayscale if img is not None: img = cv2.resize(img, (80, 140)) # Resize to 80x140 (width x height) for the model images.append(img) labels.append(label) else: LOGGER.warning(f"Failed to load image {img_path}") # Removed the 'break' statement to allow processing all labels return np.array(images), np.array(labels) # Load data LOGGER.info("Loading images from dataset") image_data, labels = load_images_from_folder("dataset/") image_data = image_data.reshape(-1, 140, 80, 1) # Add channel dimension for grayscale image_data = image_data / 255.0 # Normalize the pixel values # Convert labels to integers (e.g., ticked=0, unticked=1, circled_yes=2, circled_no=3) LOGGER.info("Labeling images") label_mapping = {"ticked": 0, "unticked": 1, "circled_yes": 2, "circled_no": 3} labels = np.array([label_mapping[label] for label in labels]) # Count the number of images per label unique_labels, counts = np.unique(labels, return_counts=True) label_names = {v: k for k, v in label_mapping.items()} # Reverse mapping print("\nNumber of images per label in the entire dataset:") for label_int, count in zip(unique_labels, counts): label_name = label_names[label_int] print(f"{label_name} ({label_int}): {count} images") # Split data into training, validation, and testing LOGGER.info("Splitting data into training, validation, and testing") X_train_full, X_test, y_train_full, y_test = train_test_split(image_data, labels, test_size=0.2, random_state=42) X_train, X_val, y_train, y_val = train_test_split( X_train_full, y_train_full, test_size=0.1, random_state=42 ) # 10% of training data for validation # Function to count labels def count_labels(y, dataset_name): unique_labels, counts = np.unique(y, return_counts=True) print(f"\nNumber of images per label in the {dataset_name} set:") for label_int, count in zip(unique_labels, counts): label_name = label_names[label_int] print(f"{label_name} ({label_int}): {count} images") # Count labels in each dataset count_labels(y_train, "training") count_labels(y_val, "validation") count_labels(y_test, "testing") # Compute class weights class_weights_array = class_weight.compute_class_weight(class_weight="balanced", classes=np.unique(y_train), y=y_train) class_weights = dict(enumerate(class_weights_array)) # Create the CNN model LOGGER.info("Creating CNN model with dropout") model = models.Sequential() # First convolutional block model.add(layers.Conv2D(32, (3, 3), activation="relu", input_shape=(140, 80, 1))) model.add(layers.MaxPooling2D((2, 2))) model.add(layers.Dropout(0.25)) # Dropout layer added # Second convolutional block model.add(layers.Conv2D(64, (3, 3), activation="relu")) model.add(layers.MaxPooling2D((2, 2))) model.add(layers.Dropout(0.25)) # Dropout layer added # Third convolutional block model.add(layers.Conv2D(64, (3, 3), activation="relu")) # Optional pooling layer if needed # model.add(layers.MaxPooling2D((2, 2))) model.add(layers.Dropout(0.25)) # Dropout layer added # Flatten and dense layers model.add(layers.Flatten()) model.add(layers.Dense(64, activation="relu")) model.add(layers.Dropout(0.5)) # Dropout layer added model.add(layers.Dense(4, activation="softmax")) optimizer = Adam(learning_rate=0.001) lr_scheduler = ReduceLROnPlateau(monitor="val_loss", factor=0.1, patience=3) # Compile the model LOGGER.info("Compiling the model") model.compile(optimizer=optimizer, loss="sparse_categorical_crossentropy", metrics=["accuracy"]) model.summary() # Train the model LOGGER.info("Training the model") early_stopping = EarlyStopping(monitor="val_loss", patience=5, restore_best_weights=True) history = model.fit( X_train, y_train, epochs=100, validation_data=(X_val, y_val), callbacks=[early_stopping, lr_scheduler], class_weight=class_weights, ) print("") print("") print("") print("") print("") # Plot training and validation accuracy and loss plt.figure(figsize=(12, 5)) plt.subplot(1, 2, 1) plt.plot(history.history["accuracy"], label="Training Accuracy", color="blue") plt.plot(history.history["val_accuracy"], label="Validation Accuracy", color="orange") plt.xlabel("Epoch") plt.ylabel("Accuracy") plt.ylim([0, 1]) plt.legend(loc="lower right") plt.title("Model Accuracy") plt.subplot(1, 2, 2) plt.plot(history.history["loss"], label="Training Loss", color="blue") plt.plot(history.history["val_loss"], label="Validation Loss", color="orange") plt.xlabel("Epoch") plt.ylabel("Loss") plt.legend(loc="upper right") plt.title("Model Loss") plt.tight_layout() plt.show() # Evaluate the model on test data test_loss, test_acc = model.evaluate(X_test, y_test, verbose=2) LOGGER.info(f"Test accuracy: {test_acc}") # Evaluate the model LOGGER.info("\nEvaluating the model") LOGGER.info("Classification Report") y_pred_probs = model.predict(X_test) y_pred = np.argmax(y_pred_probs, axis=1) print(classification_report(y_test, y_pred, target_names=label_mapping.keys()))
模型评估结果
初始评估结果
precision recall f1-score support ticked 0.96 0.97 0.96 69 unticked 0.97 0.95 0.96 59 circled_yes 0.99 1.00 0.99 79 circled_no 1.00 0.99 0.99 70 accuracy 0.98 277 macro avg 0.98 0.98 0.98 277 weighted avg 0.98 0.98 0.98 277
后续测试评估结果
precision recall f1-score support ticked 1.00 0.96 0.98 75 unticked 0.96 1.00 0.98 74 circled_yes 1.00 1.00 1.00 71 circled_no 1.00 1.00 1.00 64 accuracy 0.99 284 macro avg 0.99 0.99 0.99 284 weighted avg 0.99 0.99 0.99 284 INFO:__main__:Confusion Matrix: [[72 3 0 0] [ 0 74 0 0] [ 0 0 71 0] [ 0 0 0 64]]
优化建议
- 保留当前基线模型:当前模型准确率已达98%-99%,属于极高水准,无需强行压缩参数或盲目扩容数据。若需轻量化部署,建议用知识蒸馏——以当前模型为教师模型,训练更小的学生模型,而非直接裁剪现有模型参数。
- 排查新增数据质量:新增图像后性能下降,大概率是数据问题:标注错误、图像风格(分辨率/光照)与原有数据集差异大、混入无关样本。先抽样检查新增数据,确保标注准确、风格匹配,再分批少量加入训练,观察性能变化。
- 调整图像增广策略:现有增广可能破坏勾选/圈选的关键特征,建议只采用无破坏的增广方式:
- 轻微平移(上下左右不超过5像素)
- 小幅亮度/对比度调整(避免线条消失)
- 禁用旋转、翻转(方向是核心特征)
- 用
ImageDataGenerator实时增广,避免提前生成冗余数据
- 针对性优化易错类别:从混淆矩阵看,仅
ticked和unticked存在少量混淆,收集这两类的边缘样本(如模糊勾选、接近勾选的未勾选),单独做精细化训练,或微调这两类的类别权重。 - 微调训练参数:尝试将学习率降至0.0001做微调,或延长训练epochs(配合早停避免过拟合),填补最后1%-2%的准确率缺口。
- 加入特征可视化:用Grad-CAM可视化模型关注区域,确认模型聚焦于勾选/圈选线条而非无关区域,若关注区域偏离,再调整卷积层设计。
内容的提问来源于stack exchange,提问作者bruvio
相关产品推荐
相关产品推荐

