You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用binascii.hexlify处理后的数据切片异常问题求助

二进制存档解析问题:切片导致数据错误的解决方法

问题场景

读取某程序导出的二进制存档文件,使用binascii.hexlify处理后提取数据用于Qt GUI展示,但通过Inventory类切片后得到的数据完全不符合预期,直接用绝对偏移调用read_value却能得到正确结果。

原代码

import binascii, functools, struct
from pathlib import Path

def read_value(dataset, start, length, encoding):
    unhex = binascii.unhexlify(dataset[start * 2:(start + length) * 2])
    unpack = struct.unpack(encoding, unhex)
    data = functools.reduce(lambda result, decrypted: result * 10 + decrypted, unpack)

    return data

class Inventory:
    def __init__(self, dataset, category):
        self.dataset = dataset
        self.category = category

    def get_data(self):
        if self.category == 'collectibles':
            data = self.dataset[0x55060:0x56448]

        return data

class Item:
    def __init__(self, data, offset):
        self.item_id = read_value(data, offset + 0x0, 0x2, '<H')
        self.index = read_value(data, offset + 0x2, 0x2, '<H')
        self.category = read_value(data, offset + 0x4, 0x1, '<B')
        self.amount = read_value(data, offset + 0xC, 0x2, '<H')
        self.status = read_value(data, offset + 0xE, 0x2, '<H')

the_file = Path('../file.sav').read_bytes()
the_dataset = binascii.hexlify(the_file)
the_collectibles = Inventory(the_dataset, 'collectibles').get_data()
the_items = [Item(the_collectibles, 0x0), Item(the_collectibles, 0x10)]

print(the_items[0].__dict__)
print(the_items[1].__dict__)

预期输出

{'item_id': 2002, 'index': 0, 'category': 3, 'amount': 99, 'status': 1}
{'item_id': 2059, 'index': 1, 'category': 3, 'amount': 45, 'status': 1}

实际输出

{'item_id': 65535, 'index': 45928, 'category': 10, 'amount': 65150, 'status': 4614}
{'item_id': 65535, 'index': 0, 'category': 181, 'amount': 20489, 'status': 8093}

正确的直接调用方式

item_id = read_value(the_dataset, 0x55060, 0x2, '<H')
index = read_value(the_dataset, 0x55062, 0x2, '<H')
category = read_value(the_dataset, 0x55064, 0x1, '<B')
amount = read_value(the_dataset, 0x5506C, 0x2, '<H')
status = read_value(the_dataset, 0x5506E, 0x2, '<H')

问题原因

核心错误是对hexlify处理后的字节串长度逻辑理解错误:

  • hexlify会把原始二进制的每个字节转换成2个十六进制字符,所以处理后的字节串长度是原始数据的2倍
  • Inventory.get_data中直接用原始字节偏移0x55060:0x56448切片,取到的是错误的十六进制字符范围
  • 同时read_value函数对切片后的子串仍使用start * 2计算偏移,相当于在已经翻倍的长度上再次翻倍,进一步加剧错误

解决方法

思路1:直接操作原始二进制数据(推荐)

跳过hexlify转换,直接对原始字节数据解析,偏移计算更直观,避免字符串长度翻倍的问题:

import functools, struct
from pathlib import Path

def read_value(dataset, start, length, encoding):
    unpack = struct.unpack(encoding, dataset[start:start+length])
    data = functools.reduce(lambda result, decrypted: result * 10 + decrypted, unpack)
    return data

class Inventory:
    def __init__(self, dataset, category):
        self.dataset = dataset
        self.category = category

    def get_data(self):
        if self.category == 'collectibles':
            return self.dataset[0x55060:0x56448]
        return b''

class Item:
    def __init__(self, data, offset):
        self.item_id = read_value(data, offset + 0x0, 0x2, '<H')
        self.index = read_value(data, offset + 0x2, 0x2, '<H')
        self.category = read_value(data, offset + 0x4, 0x1, '<B')
        self.amount = read_value(data, offset + 0xC, 0x2, '<H')
        self.status = read_value(data, offset + 0xE, 0x2, '<H')

the_file = Path('../file.sav').read_bytes()
the_collectibles = Inventory(the_file, 'collectibles').get_data()
the_items = [Item(the_collectibles, 0x0), Item(the_collectibles, 0x10)]

print(the_items[0].__dict__)
print(the_items[1].__dict__)

思路2:修正hexlify后的切片和偏移计算

如果必须保留hexlify处理,需要修正两个关键点:

  1. 把原始字节偏移转换为十六进制字符串偏移(乘以2)
  2. 对切片后的子串,read_value不再进行偏移翻倍计算
import binascii, functools, struct
from pathlib import Path

def read_value(dataset, start, length, encoding):
    unhex = binascii.unhexlify(dataset[start*2 : (start+length)*2])
    unpack = struct.unpack(encoding, unhex)
    data = functools.reduce(lambda result, decrypted: result * 10 + decrypted, unpack)
    return data

class Inventory:
    def __init__(self, dataset, category):
        self.dataset = dataset
        self.category = category

    def get_data(self):
        if self.category == 'collectibles':
            start = 0x55060 * 2
            end = 0x56448 * 2
            return self.dataset[start:end]
        return b''

class Item:
    def __init__(self, data, offset):
        self.item_id = read_value(data, offset + 0x0, 0x2, '<H')
        self.index = read_value(data, offset + 0x2, 0x2, '<H')
        self.category = read_value(data, offset + 0x4, 0x1, '<B')
        self.amount = read_value(data, offset + 0xC, 0x2, '<H')
        self.status = read_value(data, offset + 0xE, 0x2, '<H')

the_file = Path('../file.sav').read_bytes()
the_dataset = binascii.hexlify(the_file)
the_collectibles = Inventory(the_dataset, 'collectibles').get_data()
the_items = [Item(the_collectibles, 0x0), Item(the_collectibles, 0x10)]

print(the_items[0].__dict__)
print(the_items[1].__dict__)

内容的提问来源于stack exchange,提问作者Xeuxs

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.08 08:51:03