You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何序列化含STL容器的C++类,导出转储文件实现快速加载?

实现方案

核心思路是给MyData类实现二进制序列化/反序列化逻辑,将所有私有容器的实际数据写入转储文件,加载时直接读取二进制数据重构容器,避免重复解析原始文件。

注意:不要尝试直接将MyData对象的内存整体写入文件,STL容器内部存储的是堆内存指针,直接dump对象本身只会写入指针地址,不会写入实际数据,加载时指针会悬空导致崩溃


步骤1:扩展MyData类,增加序列化/反序列化方法

#include <fstream>
#include <vector>
#include <unordered_map>
#include <string>
#include <cstdint>
#include <stdexcept>

class MyData {
 public:
  void Load(const std::vector<std::string>& file_paths) {
    // 原有读取原始文件的逻辑保持不变
  }

  // 将对象序列化保存为转储文件
  bool SaveDump(const std::string& dump_path) const {
    std::ofstream ofs(dump_path, std::ios::binary);
    if (!ofs.is_open()) return false;

    // 写入魔术头和版本号,避免加载错误格式/旧版本的转储文件
    const char magic[4] = {'M', 'Y', 'D', 'T'};
    ofs.write(magic, 4);
    const uint32_t version = 1;
    ofs.write(reinterpret_cast<const char*>(&version), sizeof(version));

    // 序列化vector<double> a:先写元素数量,再写所有double数据
    uint32_t a_size = static_cast<uint32_t>(a.size());
    ofs.write(reinterpret_cast<const char*>(&a_size), sizeof(a_size));
    ofs.write(reinterpret_cast<const char*>(a.data()), a_size * sizeof(double));

    // 序列化unordered_map<string, double> b:先写键值对总数,再逐个写每个键值对
    uint32_t b_size = static_cast<uint32_t>(b.size());
    ofs.write(reinterpret_cast<const char*>(&b_size), sizeof(b_size));
    for (const auto& pair : b) {
      uint32_t key_len = static_cast<uint32_t>(pair.first.size());
      ofs.write(reinterpret_cast<const char*>(&key_len), sizeof(key_len));
      ofs.write(pair.first.data(), key_len);
      ofs.write(reinterpret_cast<const char*>(&pair.second), sizeof(double));
    }

    return ofs.good();
  }

  // 从转储文件反序列化生成MyData对象
  static MyData LoadDump(const std::string& dump_path) {
    std::ifstream ifs(dump_path, std::ios::binary);
    if (!ifs.is_open()) {
      throw std::runtime_error("Failed to open dump file");
    }

    // 校验魔术头和版本号
    char magic[4];
    ifs.read(magic, 4);
    if (magic[0] != 'M' || magic[1] != 'Y' || magic[2] != 'D' || magic[3] != 'T') {
      throw std::runtime_error("Invalid dump file format");
    }
    uint32_t version;
    ifs.read(reinterpret_cast<char*>(&version), sizeof(version));
    if (version != 1) {
      throw std::runtime_error("Unsupported dump file version");
    }

    MyData res;

    // 反序列化vector<double> a
    uint32_t a_size;
    ifs.read(reinterpret_cast<char*>(&a_size), sizeof(a_size));
    res.a.resize(a_size);
    ifs.read(reinterpret_cast<char*>(res.a.data()), a_size * sizeof(double));

    // 反序列化unordered_map<string, double> b
    uint32_t b_size;
    ifs.read(reinterpret_cast<char*>(&b_size), sizeof(b_size));
    res.b.reserve(b_size); // 预分配空间提升加载速度
    for (uint32_t i = 0; i < b_size; ++i) {
      uint32_t key_len;
      ifs.read(reinterpret_cast<char*>(&key_len), sizeof(key_len));
      std::string key(key_len, '\0');
      ifs.read(key.data(), key_len);
      double val;
      ifs.read(reinterpret_cast<char*>(&val), sizeof(val));
      res.b.emplace(std::move(key), val);
    }

    if (!ifs.good()) {
      throw std::runtime_error("Failed to read dump file, data may be corrupted");
    }
    return res;
  }

 private:
  std::vector<double> a;
  std::unordered_map<std::string, double> b;
  // 后续新增容器只需对应扩展序列化/反序列化逻辑即可
};

步骤2:封装你需要的load_dump函数

MyData load_dump(const std::string& dump_file_path) {
  return MyData::LoadDump(dump_file_path);
}

使用方式

// 首次使用(原始文件更新时)生成转储文件
MyData mdata;
std::vector<std::string> fs; // 数千个原始文件路径
mdata.Load(fs);
mdata.SaveDump("dump.bin");

// 后续日常使用直接加载转储文件,速度提升可达几十上百倍
const MyData& mdata = load_dump("dump.bin");

注意事项

  • 上述实现默认在同架构同平台使用,如果需要跨操作系统/跨CPU架构使用,需要额外处理大小端、类型长度对齐问题,所有数值统一转成小端/大端存储,读取时再转成本地字节序。
  • 后续如果新增成员容器,只需在SaveDump和LoadDump方法中对应新增容器的读写逻辑,同步升级版本号即可避免旧版本转储加载出错。
  • 数据量极大的情况下可以引入zlib等压缩库对二进制流做压缩,减少磁盘占用。

内容的提问来源于stack exchange,提问作者nick

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.27 06:36:02