You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Boost序列化按批次读取二进制文件时崩溃问题排查

问题分析与解决方案

你遇到的崩溃问题核心原因是Boost二进制归档(binary_iarchive)的工作机制和你操作文件指针的方式不兼容,具体有两个关键错误:

  1. 每次调用load时,你先创建binary_iarchive再移动文件指针——但binary_iarchive在构造时就会读取文件开头的归档头信息(魔术数字、版本号等),这会让文件指针自动前进到归档头之后的位置。后续手动移动指针的操作会让归档的内部状态和实际文件位置完全脱节,导致读取时解析错误崩溃。
  2. 第二次调用load时,你把文件指针移到第一次读完5条记录的位置后创建binary_iarchive——这时候归档会尝试从当前位置读取归档头,但这里已经是记录数据了,根本不是合法的归档头,直接触发崩溃。

下面给出两种可行的解决方案:

方案一:持久化文件流与归档对象(推荐)

这种方式保持文件流和归档对象在整个程序生命周期中打开,避免每次重新创建归档时重复读取归档头,同时归档会自动维护读取位置,不需要手动操作文件指针。

修改后的代码如下:

#include <boost/archive/binary_iarchive.hpp>
#include <boost/archive/binary_oarchive.hpp>
#include <boost/serialization/string.hpp>
#include <fstream>
#include <iostream>
#include <unistd.h> // sleep函数需要该头文件
#include <algorithm>
using namespace std;
using namespace boost::archive;

class logEntry {
private:
    size_t m_txID;
    string m_jsonStr;
    friend class boost::serialization::access;
    template <typename Archive>
    friend void serialize( Archive &ar, logEntry &l, const unsigned int version );
public:
    logEntry() { m_txID = 0; m_jsonStr = ""; }
    logEntry( size_t id, const string &val ) { m_txID = id; m_jsonStr = val; }
    string getJsonValue() { return m_jsonStr; }
    size_t getTxId() { return m_txID; }
};

template <typename Archive>
void serialize( Archive &ar, logEntry &l, const unsigned int version ) {
    ar &l.m_txID;
    ar &l.m_jsonStr;
}

// 全局持久化的文件流和归档对象
ifstream* file = nullptr;
binary_iarchive* ia = nullptr;
size_t totalRecords = 10;
size_t recordsRead = 0;

void save() {
    ofstream file{"/tmp/test.bin", ios::binary | ios::trunc};
    binary_oarchive oa{file};
    // 先写入总记录数,方便读取时判断剩余量
    oa << totalRecords;
    // 保存10条记录
    for ( int i = 0; i < totalRecords; i++ )
        oa << logEntry( i, "{Some Json String}" );
    file.flush();
    file.close();
}

// 分批加载数据
void load( size_t bsize ) {
    // 第一次调用时初始化文件流和归档
    if (!file) {
        file = new ifstream{"/tmp/test.bin", ios::binary};
        ia = new binary_iarchive{*file};
        // 读取总记录数
        *ia >> totalRecords;
    }

    logEntry l;
    size_t recordsToRead = min(bsize, totalRecords - recordsRead);
    for ( size_t i = 0; i < recordsToRead; i++ ) {
        *ia >> l;
        // 这里可以添加记录处理逻辑,比如打印验证
        cout << "读取记录:txID=" << l.getTxId() << ", json=" << l.getJsonValue() << endl;
        recordsRead++;
    }

    // 所有记录读完后释放资源并重置
    if (recordsRead >= totalRecords) {
        delete ia;
        delete file;
        ia = nullptr;
        file = nullptr;
        recordsRead = 0;
        cout << "所有记录读取完成,重置状态..." << endl;
    }
}

int main() {
    save();
    while ( 1 ) {
        load( 5 );
        sleep( 5 );
    }
}

这个方案高效可靠,归档只初始化一次,不需要手动管理文件指针,归档会自动跟踪读取位置。

方案二:每次重新打开文件,跳过已读取的记录

如果必须每次调用load都重新打开文件(比如需要程序重启后恢复读取位置),可以在保存时记录总记录数,每次读取时先创建归档,跳过已读记录后再读取当前批次,同时将已读记录数持久化到临时文件。

示例代码:

#include <boost/archive/binary_iarchive.hpp>
#include <boost/archive/binary_oarchive.hpp>
#include <boost/serialization/string.hpp>
#include <fstream>
#include <iostream>
#include <unistd.h>
#include <algorithm>
using namespace std;
using namespace boost::archive;

class logEntry {
private:
    size_t m_txID;
    string m_jsonStr;
    friend class boost::serialization::access;
    template <typename Archive>
    friend void serialize( Archive &ar, logEntry &l, const unsigned int version );
public:
    logEntry() { m_txID = 0; m_jsonStr = ""; }
    logEntry( size_t id, const string &val ) { m_txID = id; m_jsonStr = val; }
    string getJsonValue() { return m_jsonStr; }
    size_t getTxId() { return m_txID; }
};

template <typename Archive>
void serialize( Archive &ar, logEntry &l, const unsigned int version ) {
    ar &l.m_txID;
    ar &l.m_jsonStr;
}

const string posFile = "/tmp/read_pos.txt";

// 保存已读取的记录数
void saveReadPos(size_t count) {
    ofstream f(posFile);
    f << count;
}

// 加载已读取的记录数
size_t loadReadPos() {
    ifstream f(posFile);
    size_t count = 0;
    if (f.is_open()) {
        f >> count;
    }
    return count;
}

void save() {
    ofstream file{"/tmp/test.bin", ios::binary | ios::trunc};
    binary_oarchive oa{file};
    size_t totalRecords = 10;
    oa << totalRecords;
    for ( int i = 0; i < totalRecords; i++ )
        oa << logEntry( i, "{Some Json String}" );
    file.flush();
    file.close();
    // 重置读取位置
    saveReadPos(0);
}

void load( size_t bsize ) {
    ifstream file{"/tmp/test.bin", ios::binary};
    binary_iarchive ia{file};

    size_t totalRecords;
    ia >> totalRecords;

    size_t recordsRead = loadReadPos();
    logEntry l;

    // 跳过已读取的记录
    for (size_t i = 0; i < recordsRead; i++) {
        ia >> l;
    }

    // 读取当前批次
    size_t recordsToRead = min(bsize, totalRecords - recordsRead);
    for ( size_t i = 0; i < recordsToRead; i++ ) {
        ia >> l;
        cout << "读取记录:txID=" << l.getTxId() << ", json=" << l.getJsonValue() << endl;
        recordsRead++;
    }

    saveReadPos(recordsRead);
    file.close();

    if (recordsRead >= totalRecords) {
        cout << "所有记录读取完成,重置位置..." << endl;
        saveReadPos(0);
    }
}

int main() {
    save();
    while ( 1 ) {
        load( 5 );
        sleep( 5 );
    }
}

这个方案支持程序重启后恢复读取进度,适合需要断点续读的场景,唯一的小缺点是每次重新读取时需要跳过已读记录,效率略低于方案一。

关键注意事项

  • Boost归档对象和对应流强绑定,归档内部维护了序列化状态(如已处理字节数、元数据),不要手动修改流位置后再使用归档对象,否则会导致状态不一致引发崩溃。
  • 分批读取时,要么保持归档和流的持久化,要么通过跳过已读记录实现,避免直接操作文件指针。
  • 建议保存数据时先写入总记录数,这样读取时可以明确剩余记录量,简化文件结尾判断逻辑。

内容的提问来源于stack exchange,提问作者user3620473

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.15 08:52:09