使用binascii.hexlify处理后的数据切片异常问题求助
二进制存档解析问题:切片导致数据错误的解决方法
问题场景
读取某程序导出的二进制存档文件,使用binascii.hexlify处理后提取数据用于Qt GUI展示,但通过Inventory类切片后得到的数据完全不符合预期,直接用绝对偏移调用read_value却能得到正确结果。
原代码
import binascii, functools, struct from pathlib import Path def read_value(dataset, start, length, encoding): unhex = binascii.unhexlify(dataset[start * 2:(start + length) * 2]) unpack = struct.unpack(encoding, unhex) data = functools.reduce(lambda result, decrypted: result * 10 + decrypted, unpack) return data class Inventory: def __init__(self, dataset, category): self.dataset = dataset self.category = category def get_data(self): if self.category == 'collectibles': data = self.dataset[0x55060:0x56448] return data class Item: def __init__(self, data, offset): self.item_id = read_value(data, offset + 0x0, 0x2, '<H') self.index = read_value(data, offset + 0x2, 0x2, '<H') self.category = read_value(data, offset + 0x4, 0x1, '<B') self.amount = read_value(data, offset + 0xC, 0x2, '<H') self.status = read_value(data, offset + 0xE, 0x2, '<H') the_file = Path('../file.sav').read_bytes() the_dataset = binascii.hexlify(the_file) the_collectibles = Inventory(the_dataset, 'collectibles').get_data() the_items = [Item(the_collectibles, 0x0), Item(the_collectibles, 0x10)] print(the_items[0].__dict__) print(the_items[1].__dict__)
预期输出
{'item_id': 2002, 'index': 0, 'category': 3, 'amount': 99, 'status': 1} {'item_id': 2059, 'index': 1, 'category': 3, 'amount': 45, 'status': 1}
实际输出
{'item_id': 65535, 'index': 45928, 'category': 10, 'amount': 65150, 'status': 4614} {'item_id': 65535, 'index': 0, 'category': 181, 'amount': 20489, 'status': 8093}
正确的直接调用方式
item_id = read_value(the_dataset, 0x55060, 0x2, '<H') index = read_value(the_dataset, 0x55062, 0x2, '<H') category = read_value(the_dataset, 0x55064, 0x1, '<B') amount = read_value(the_dataset, 0x5506C, 0x2, '<H') status = read_value(the_dataset, 0x5506E, 0x2, '<H')
问题原因
核心错误是对hexlify处理后的字节串长度逻辑理解错误:
hexlify会把原始二进制的每个字节转换成2个十六进制字符,所以处理后的字节串长度是原始数据的2倍Inventory.get_data中直接用原始字节偏移0x55060:0x56448切片,取到的是错误的十六进制字符范围- 同时
read_value函数对切片后的子串仍使用start * 2计算偏移,相当于在已经翻倍的长度上再次翻倍,进一步加剧错误
解决方法
思路1:直接操作原始二进制数据(推荐)
跳过hexlify转换,直接对原始字节数据解析,偏移计算更直观,避免字符串长度翻倍的问题:
import functools, struct from pathlib import Path def read_value(dataset, start, length, encoding): unpack = struct.unpack(encoding, dataset[start:start+length]) data = functools.reduce(lambda result, decrypted: result * 10 + decrypted, unpack) return data class Inventory: def __init__(self, dataset, category): self.dataset = dataset self.category = category def get_data(self): if self.category == 'collectibles': return self.dataset[0x55060:0x56448] return b'' class Item: def __init__(self, data, offset): self.item_id = read_value(data, offset + 0x0, 0x2, '<H') self.index = read_value(data, offset + 0x2, 0x2, '<H') self.category = read_value(data, offset + 0x4, 0x1, '<B') self.amount = read_value(data, offset + 0xC, 0x2, '<H') self.status = read_value(data, offset + 0xE, 0x2, '<H') the_file = Path('../file.sav').read_bytes() the_collectibles = Inventory(the_file, 'collectibles').get_data() the_items = [Item(the_collectibles, 0x0), Item(the_collectibles, 0x10)] print(the_items[0].__dict__) print(the_items[1].__dict__)
思路2:修正hexlify后的切片和偏移计算
如果必须保留hexlify处理,需要修正两个关键点:
- 把原始字节偏移转换为十六进制字符串偏移(乘以2)
- 对切片后的子串,
read_value不再进行偏移翻倍计算
import binascii, functools, struct from pathlib import Path def read_value(dataset, start, length, encoding): unhex = binascii.unhexlify(dataset[start*2 : (start+length)*2]) unpack = struct.unpack(encoding, unhex) data = functools.reduce(lambda result, decrypted: result * 10 + decrypted, unpack) return data class Inventory: def __init__(self, dataset, category): self.dataset = dataset self.category = category def get_data(self): if self.category == 'collectibles': start = 0x55060 * 2 end = 0x56448 * 2 return self.dataset[start:end] return b'' class Item: def __init__(self, data, offset): self.item_id = read_value(data, offset + 0x0, 0x2, '<H') self.index = read_value(data, offset + 0x2, 0x2, '<H') self.category = read_value(data, offset + 0x4, 0x1, '<B') self.amount = read_value(data, offset + 0xC, 0x2, '<H') self.status = read_value(data, offset + 0xE, 0x2, '<H') the_file = Path('../file.sav').read_bytes() the_dataset = binascii.hexlify(the_file) the_collectibles = Inventory(the_dataset, 'collectibles').get_data() the_items = [Item(the_collectibles, 0x0), Item(the_collectibles, 0x10)] print(the_items[0].__dict__) print(the_items[1].__dict__)
内容的提问来源于stack exchange,提问作者Xeuxs
相关产品推荐
相关产品推荐

