You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

实现MidiReader:如何优雅读取不同字节序的UInt类型

实现MidiReader的无符号整数读取方案

我正在实现一个MidiReader,需要读取大端(MSB First)或小端(LSB First)的8、16、32、64位无符号整数(UInt)。由于对二进制和类型知识了解较少,目前参考C#代码编写了C++的ByteArrayReader类及ReadUInt16、ReadUInt32方法:

class ByteArrayReader
{
public:
    unsigned char* ByteArray;
    unsigned int Size;
    unsigned int Index = 0;

    ByteArrayReader(unsigned char* byteArray)
    {
        if (byteArray == NULL)
        {
            throw byteArray;
        }
        ByteArray = byteArray;
        Size = (unsigned int)sizeof(byteArray);
        Index = 0;
    }

    char inline Read()
    {
        return ByteArray[Index++];
    }

    void inline Forward(unsigned int length = 1)
    {
        Index += length;
    }

    void inline Backward(unsigned int length = 1)
    {
        if (length > Index)
        {
            throw length;
        }

        Index -= length;
    }

    bool operator==(ByteArrayReader) = delete;
};
uint16_t inline ReadUInt16()
{
    return (uint16_t)((Read() << 8) | Read());
}

uint32_t inline ReadUInt32()
{
    return (uint32_t)((((((Read() << 8) | Read()) << 8) | Read()) << 8) | Read());
}

已知上述方法是读取大端UInt,想了解如何优雅地从二进制中读取UInt类型,同时学习UInt的字节表示方式。


一、UInt的字节表示逻辑

  • 大端(MSB First):最高有效字节(数值权重最大的字节)存在内存的最低地址。比如32位UInt值0x12345678,大端存储顺序为0x12 → 0x34 → 0x56 → 0x78(地址从低到高),你的现有ReadUInt16/ReadUInt32就是按这个逻辑实现的。
  • 小端(LSB First):最低有效字节(数值权重最小的字节)存在内存的最低地址。同样0x12345678,小端存储顺序为0x78 → 0x56 → 0x34 → 0x12。

二、现有代码的关键bug修复

你的ByteArrayReader构造函数存在致命问题:Size = (unsigned int)sizeof(byteArray);中,byteArray是指针,sizeof返回的是指针本身的大小(64位系统为8字节),而非传入数组的实际长度。必须修改为让调用者主动传入数组大小:

ByteArrayReader(unsigned char* byteArray, unsigned int size)
{
    if (byteArray == nullptr)
    {
        throw std::invalid_argument("byteArray cannot be null");
    }
    ByteArray = byteArray;
    Size = size;
    Index = 0;
}

另外Read()返回char可能触发符号扩展问题,应改为返回unsigned char,同时增加越界检查:

unsigned char inline Read()
{
    if (Index >= Size)
        throw std::out_of_range("Reader out of bounds");
    return ByteArray[Index++];
}

三、优雅的跨端序读取实现

1. 通用模板方法(适配所有无符号整数类型)

通过模板函数统一处理不同长度的UInt,用枚举指定端序,避免重复编写相似逻辑:

#include <cstdint>
#include <stdexcept>
#include <type_traits>

// 定义端序枚举
enum class Endianness {
    Big,
    Little
};

class ByteArrayReader
{
public:
    unsigned char* ByteArray;
    unsigned int Size;
    unsigned int Index = 0;

    ByteArrayReader(unsigned char* byteArray, unsigned int size)
        : ByteArray(byteArray), Size(size)
    {
        if (!byteArray)
            throw std::invalid_argument("Null byte array");
    }

    unsigned char inline Read()
    {
        if (Index >= Size)
            throw std::out_of_range("Reader overflow");
        return ByteArray[Index++];
    }

    void inline Forward(unsigned int length = 1)
    {
        if (Index + length > Size)
            throw std::out_of_range("Forward overflow");
        Index += length;
    }

    void inline Backward(unsigned int length = 1)
    {
        if (length > Index)
            throw std::out_of_range("Backward underflow");
        Index -= length;
    }

    // 通用无符号整数读取模板
    template<typename T>
    T ReadUInt(Endianness endian)
    {
        static_assert(std::is_unsigned_v<T>, "T must be an unsigned integer type");
        constexpr size_t byteCount = sizeof(T);

        if (Index + byteCount > Size)
            throw std::out_of_range("Not enough bytes to read");

        T result = 0;
        if (endian == Endianness::Big)
        {
            for (size_t i = 0; i < byteCount; ++i)
            {
                result = (result << 8) | Read();
            }
        }
        else
        {
            for (size_t i = 0; i < byteCount; ++i)
            {
                result |= static_cast<T>(Read()) << (i * 8);
            }
        }
        return result;
    }

    // 快捷方法:大端读取
    uint8_t ReadUInt8() { return ReadUInt<uint8_t>(Endianness::Big); }
    uint16_t ReadUInt16Big() { return ReadUInt<uint16_t>(Endianness::Big); }
    uint32_t ReadUInt32Big() { return ReadUInt<uint32_t>(Endianness::Big); }
    uint64_t ReadUInt64Big() { return ReadUInt<uint64_t>(Endianness::Big); }

    // 快捷方法:小端读取
    uint16_t ReadUInt16Little() { return ReadUInt<uint16_t>(Endianness::Little); }
    uint32_t ReadUInt32Little() { return ReadUInt<uint32_t>(Endianness::Little); }
    uint64_t ReadUInt64Little() { return ReadUInt<uint64_t>(Endianness::Little); }

    bool operator==(ByteArrayReader) = delete;
};

2. C++20+标准库优化方案

如果编译器支持C++20,可以用<bit>头文件的std::endian和std::byteswap实现更高效的端序转换:

#include <bit>
#include <cstring>

// 读取原始字节后根据端序转换
template<typename T>
T ReadUInt(Endianness endian)
{
    static_assert(std::is_unsigned_v<T>, "T must be unsigned");
    constexpr size_t byteCount = sizeof(T);

    if (Index + byteCount > Size)
        throw std::out_of_range("Overflow");

    T result;
    std::memcpy(&result, ByteArray + Index, byteCount);
    Index += byteCount;

    // 仅当目标端序与系统原生端序不同时才转换
    if ((endian == Endianness::Big && std::endian::native == std::endian::little) ||
        (endian == Endianness::Little && std::endian::native == std::endian::big))
    {
        result = std::byteswap(result);
    }
    return result;
}

四、使用示例

int main()
{
    unsigned char data[] = {0x12, 0x34, 0x56, 0x78};
    ByteArrayReader reader(data, sizeof(data));

    // 大端读取32位整数,结果为0x12345678 = 305419896
    uint32_t bigUInt = reader.ReadUInt32Big();
    // 重置索引
    reader.Backward(4);
    // 小端读取32位整数,结果为0x78563412 = 2018915346
    uint32_t littleUInt = reader.ReadUInt32Little();

    return 0;
}

内容的提问来源于stack exchange,提问作者Player01

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.04 00:56:07