You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何修复图像隐写程序的Unicode字符嵌入与提取问题?

问题描述

已实现图像数据嵌入功能,但仅支持拉丁字符及带变音符号的字符,无法处理西里尔文、希腊文、阿拉伯文、中文、日文汉字等Unicode字符,嵌入后提取结果为乱码。尝试将字符编码从8位('08b')改为16位('16b')后,问题仍未解决。需在12月5日前完成修复。

原程序代码

def embed_text(self, image_path, text, output_path):
    # Convert text message to binary format
    binary_message = ''.join(format(ord(char), '08b') for char in text) #aligning the bytes for embedding the message 
    # Load the image
    image = Image.open(image_path)
    w, h = image.size

    # Calculate the number of embedding characters (eN)
    eN = (h * w * 3) // 8
    if len(text) > eN:
        raise ValueError("Message too long to fit in the image")

    # Embedding loop
    message_index = 0
    for i in range(h):
        for j in range(w):
            pixel = list(image.getpixel((j, i)))

            for k in range(3):  # For R, G, B components
                if message_index < len(binary_message):
                    M = int(binary_message[message_index])
                    # Perform XOR operation with the 7th bit of the RGB component
                    pixel[k] = (pixel[k] & 0xFE) | (((pixel[k] >> 1) & 1) ^ M)
                    message_index += 1
                else:
                    break  # No more message bits to embed

            image.putpixel((j, i), tuple(pixel))

    # Save the Stego Image
    image.save(output_path)

def xor_substitution(self, component, bit):
    # Perform XOR on the least significant bit of the component with the bit
    return (component & 0xFE) | (component & 1) ^ bit

def extract_text(self, image_path):
    stego_image = Image.open(image_path)
    w, h = stego_image.size
    binary_message = ""

    for i in range(h):
        for j in range(w):
            pixel = stego_image.getpixel((j, i))
            for k in range(3):
                binary_message += str(pixel[k] & 1)

    # Extract only up to the NULL character
    end = binary_message.find('00000000')
    if end != -1:
        binary_message = binary_message[:end]

    return self.binary_to_string(binary_message)

def binary_to_string(self, binary_message):
    text = ""
    for i in range(0, len(binary_message), 8):
        byte = binary_message[i:i+8]
        text += chr(int(byte, 2))
    return text

修改后的代码

def embed_text(self, image_path, text, output_path):
    # Convert text message to binary format
    binary_message = ''.join(format(ord(char), '16b') for char in text)

    # Load the image
    image = Image.open(image_path)
    w, h = image.size

    # Calculate the number of embedding characters (eN)
    eN = (h * w * 3) // 16
    if len(text) > eN:
        raise ValueError("Message too long to fit in the image")
    binary_message = binary_message.ljust(eN * 16, '0')


    # Embedding loop
    message_index = 0
    for i in range(h):
        for j in range(w):
            pixel = list(image.getpixel((j, i)))

            for k in range(3):  # For R, G, B components
                if message_index < len(binary_message):
                    M = int(binary_message[message_index:message_index+16], 2)
                    # Perform XOR operation with the 7th bit of the RGB component
                    pixel[k] = (pixel[k] & 0xFFFE) | (((pixel[k] >> 1) & 1) ^ M)
                    message_index += 16
                else:
                    break  # No more message bits to embed

            image.putpixel((j, i), tuple(pixel))

    # Save the Stego Image
    image.save(output_path)

def xor_substitution(self, component, bit):
    # Perform XOR on the least significant bit of the component with the bit
    return (component & 0xFE) | (component & 1) ^ bit

def extract_text(self, image_path):
    stego_image = Image.open(image_path)
    w, h = stego_image.size
    binary_message = ""

    for i in range(h):
        for j in range(w):
            pixel = stego_image.getpixel((j, i))
            for k in range(3):
                binary_message += format(pixel[k] & 1, 'b').zfill(16)[-1]

    # Extract only up to the NULL character
    end = binary_message.find('00000000')
    if end != -1:
        binary_message = binary_message[:end]

    return self.binary_to_string(binary_message)

def binary_to_string(self, binary_message):
    text = ""
    for i in range(0, len(binary_message), 16):
        byte = binary_message[i:i+16]
        text += chr(int(byte, 2))
    return text

修复方案

核心问题分析

  1. 编码逻辑错误:直接用ord(char)转换二进制无法适配所有Unicode字符(部分字符需4字节存储),且修改后的代码错误地尝试将16位数据塞进单个像素通道的1位中,导致数据完全丢失。
  2. 嵌入逻辑混乱:单个像素通道最低位仅能存储1位二进制,一次性写入16位的操作完全不符合LSB隐写的基本规则。
  3. 终止符不匹配:修改为16位编码后,仍使用8位空字符作为终止符,导致提取时截断位置错误。

修复后完整代码

from PIL import Image

class Steganography:
    def embed_text(self, image_path, text, output_path):
        # 将文本转为UTF-8字节流,再转为二进制字符串
        byte_message = text.encode('utf-8')
        binary_message = ''.join(format(byte, '08b') for byte in byte_message)
        # 添加UTF-8空字符作为终止符(8位0)
        binary_message += '00000000'

        # 加载图像
        image = Image.open(image_path)
        w, h = image.size
        total_bits = h * w * 3
        if len(binary_message) > total_bits:
            raise ValueError("消息过长,无法嵌入到图像中")

        # 逐位嵌入到像素最低位
        message_index = 0
        for i in range(h):
            for j in range(w):
                pixel = list(image.getpixel((j, i)))
                for k in range(3):
                    if message_index < len(binary_message):
                        bit = int(binary_message[message_index])
                        pixel[k] = (pixel[k] & 0xFE) | bit
                        message_index += 1
                    else:
                        break
                image.putpixel((j, i), tuple(pixel))
                if message_index >= len(binary_message):
                    break
            if message_index >= len(binary_message):
                break

        image.save(output_path)

    def extract_text(self, image_path):
        stego_image = Image.open(image_path)
        w, h = stego_image.size
        binary_message = ""

        # 提取所有像素通道的最低位
        for i in range(h):
            for j in range(w):
                pixel = stego_image.getpixel((j, i))
                for k in range(3):
                    binary_message += str(pixel[k] & 1)
                    # 检测到终止符则停止提取
                    if len(binary_message) >= 8 and binary_message[-8:] == '00000000':
                        break
                if len(binary_message) >= 8 and binary_message[-8:] == '00000000':
                    break
            if len(binary_message) >= 8 and binary_message[-8:] == '00000000':
                break

        # 移除终止符并解码为UTF-8文本
        binary_message = binary_message[:-8]
        byte_array = bytearray()
        for i in range(0, len(binary_message), 8):
            byte = binary_message[i:i+8]
            if len(byte) == 8:
                byte_array.append(int(byte, 2))
        return byte_array.decode('utf-8', errors='replace')

关键修改说明

  • UTF-8统一编码:通过text.encode('utf-8')将所有Unicode字符转为标准字节流,避免编码长度不固定的问题。
  • 逐位嵌入:严格遵循LSB隐写规则,每个像素通道最低位仅存储1位数据,确保数据完整。
  • 终止符同步:嵌入和提取均使用UTF-8空字符作为终止符,保证消息截断准确。
  • 容错处理:解码时添加errors='replace',避免数据损坏导致程序崩溃。

内容的提问来源于stack exchange,提问作者user10116100

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.05 21:19:57