如何修复图像隐写程序的Unicode字符嵌入与提取问题?
问题描述
已实现图像数据嵌入功能,但仅支持拉丁字符及带变音符号的字符,无法处理西里尔文、希腊文、阿拉伯文、中文、日文汉字等Unicode字符,嵌入后提取结果为乱码。尝试将字符编码从8位('08b')改为16位('16b')后,问题仍未解决。需在12月5日前完成修复。
原程序代码
def embed_text(self, image_path, text, output_path): # Convert text message to binary format binary_message = ''.join(format(ord(char), '08b') for char in text) #aligning the bytes for embedding the message # Load the image image = Image.open(image_path) w, h = image.size # Calculate the number of embedding characters (eN) eN = (h * w * 3) // 8 if len(text) > eN: raise ValueError("Message too long to fit in the image") # Embedding loop message_index = 0 for i in range(h): for j in range(w): pixel = list(image.getpixel((j, i))) for k in range(3): # For R, G, B components if message_index < len(binary_message): M = int(binary_message[message_index]) # Perform XOR operation with the 7th bit of the RGB component pixel[k] = (pixel[k] & 0xFE) | (((pixel[k] >> 1) & 1) ^ M) message_index += 1 else: break # No more message bits to embed image.putpixel((j, i), tuple(pixel)) # Save the Stego Image image.save(output_path) def xor_substitution(self, component, bit): # Perform XOR on the least significant bit of the component with the bit return (component & 0xFE) | (component & 1) ^ bit def extract_text(self, image_path): stego_image = Image.open(image_path) w, h = stego_image.size binary_message = "" for i in range(h): for j in range(w): pixel = stego_image.getpixel((j, i)) for k in range(3): binary_message += str(pixel[k] & 1) # Extract only up to the NULL character end = binary_message.find('00000000') if end != -1: binary_message = binary_message[:end] return self.binary_to_string(binary_message) def binary_to_string(self, binary_message): text = "" for i in range(0, len(binary_message), 8): byte = binary_message[i:i+8] text += chr(int(byte, 2)) return text
修改后的代码
def embed_text(self, image_path, text, output_path): # Convert text message to binary format binary_message = ''.join(format(ord(char), '16b') for char in text) # Load the image image = Image.open(image_path) w, h = image.size # Calculate the number of embedding characters (eN) eN = (h * w * 3) // 16 if len(text) > eN: raise ValueError("Message too long to fit in the image") binary_message = binary_message.ljust(eN * 16, '0') # Embedding loop message_index = 0 for i in range(h): for j in range(w): pixel = list(image.getpixel((j, i))) for k in range(3): # For R, G, B components if message_index < len(binary_message): M = int(binary_message[message_index:message_index+16], 2) # Perform XOR operation with the 7th bit of the RGB component pixel[k] = (pixel[k] & 0xFFFE) | (((pixel[k] >> 1) & 1) ^ M) message_index += 16 else: break # No more message bits to embed image.putpixel((j, i), tuple(pixel)) # Save the Stego Image image.save(output_path) def xor_substitution(self, component, bit): # Perform XOR on the least significant bit of the component with the bit return (component & 0xFE) | (component & 1) ^ bit def extract_text(self, image_path): stego_image = Image.open(image_path) w, h = stego_image.size binary_message = "" for i in range(h): for j in range(w): pixel = stego_image.getpixel((j, i)) for k in range(3): binary_message += format(pixel[k] & 1, 'b').zfill(16)[-1] # Extract only up to the NULL character end = binary_message.find('00000000') if end != -1: binary_message = binary_message[:end] return self.binary_to_string(binary_message) def binary_to_string(self, binary_message): text = "" for i in range(0, len(binary_message), 16): byte = binary_message[i:i+16] text += chr(int(byte, 2)) return text
修复方案
核心问题分析
- 编码逻辑错误:直接用
ord(char)转换二进制无法适配所有Unicode字符(部分字符需4字节存储),且修改后的代码错误地尝试将16位数据塞进单个像素通道的1位中,导致数据完全丢失。 - 嵌入逻辑混乱:单个像素通道最低位仅能存储1位二进制,一次性写入16位的操作完全不符合LSB隐写的基本规则。
- 终止符不匹配:修改为16位编码后,仍使用8位空字符作为终止符,导致提取时截断位置错误。
修复后完整代码
from PIL import Image class Steganography: def embed_text(self, image_path, text, output_path): # 将文本转为UTF-8字节流,再转为二进制字符串 byte_message = text.encode('utf-8') binary_message = ''.join(format(byte, '08b') for byte in byte_message) # 添加UTF-8空字符作为终止符(8位0) binary_message += '00000000' # 加载图像 image = Image.open(image_path) w, h = image.size total_bits = h * w * 3 if len(binary_message) > total_bits: raise ValueError("消息过长,无法嵌入到图像中") # 逐位嵌入到像素最低位 message_index = 0 for i in range(h): for j in range(w): pixel = list(image.getpixel((j, i))) for k in range(3): if message_index < len(binary_message): bit = int(binary_message[message_index]) pixel[k] = (pixel[k] & 0xFE) | bit message_index += 1 else: break image.putpixel((j, i), tuple(pixel)) if message_index >= len(binary_message): break if message_index >= len(binary_message): break image.save(output_path) def extract_text(self, image_path): stego_image = Image.open(image_path) w, h = stego_image.size binary_message = "" # 提取所有像素通道的最低位 for i in range(h): for j in range(w): pixel = stego_image.getpixel((j, i)) for k in range(3): binary_message += str(pixel[k] & 1) # 检测到终止符则停止提取 if len(binary_message) >= 8 and binary_message[-8:] == '00000000': break if len(binary_message) >= 8 and binary_message[-8:] == '00000000': break if len(binary_message) >= 8 and binary_message[-8:] == '00000000': break # 移除终止符并解码为UTF-8文本 binary_message = binary_message[:-8] byte_array = bytearray() for i in range(0, len(binary_message), 8): byte = binary_message[i:i+8] if len(byte) == 8: byte_array.append(int(byte, 2)) return byte_array.decode('utf-8', errors='replace')
关键修改说明
- UTF-8统一编码:通过
text.encode('utf-8')将所有Unicode字符转为标准字节流,避免编码长度不固定的问题。 - 逐位嵌入:严格遵循LSB隐写规则,每个像素通道最低位仅存储1位数据,确保数据完整。
- 终止符同步:嵌入和提取均使用UTF-8空字符作为终止符,保证消息截断准确。
- 容错处理:解码时添加
errors='replace',避免数据损坏导致程序崩溃。
内容的提问来源于stack exchange,提问作者user10116100
相关产品推荐
相关产品推荐

