You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

二进制文件导航异常:构建索引后无法按ID读取文本内容

问题描述

现有文本文件poem.txt,内容为:

10some information#11some more information#12and some more information#

其中10、11、12是固定占2字节的句子ID。需要构建一个二进制索引文件,每条条目包含句子ID(2字节)、句子起始位置(int)、字符串长度(int),用于后续按ID读取对应句子。目前索引文件已生成,但按ID查询时始终提示“不存在该ID”,附上C++代码请求排查。

原代码:

#include <iostream>
#include <locale>
#include <string.h>

using namespace std;


int main()
{
    FILE *poem, *index;
    int N = 4096, len;
    char *str= new char[N];
    if (!(poem = fopen("poem.txt", "r")))
    {
        cout << "File not found";
        return 0;
    }
    fgets(str, N, poem);

index = fopen("index.txt", "w+b");

int pos1 = 2, pos_b;
char* pos = str+pos1, *pstr=str;
fwrite(pstr, sizeof(char), 2, index);
fwrite(&pos1, sizeof(int), 1, index);
for (int i = 2; *pstr != '\0'; ++i, pstr++) {
    if (*pstr == '#') {
        len = pstr - pos;
        fwrite(&len, sizeof(int), 1, index);
        pos = pstr + pos1;
        if (*(pstr+1)!='\0') {
           fputs("\n", index);
           fwrite(pstr+1, sizeof(char), 2, index);
           pos_b = i + 3;
           fwrite(&pos_b, sizeof(int), 1, index);
        }  
    }
}

pos_b = 0;
len = 0;
fseek(index, 0, SEEK_SET);
fseek(poem, 0, SEEK_SET);
char* id = new char[3], * id_f = new char[3], * sent = new char[N];
int have_id = 0;
cout << "Enter the id of the offer you want to read" << endl;
cin >> id;

while (!feof(index)) {
    fread(id_f, sizeof(char), 2, index);
     if (strcmp(id_f,id)==0) { 
        fread(&pos_b, sizeof(int), 1, index);
        fread(&len, sizeof(int), 1, index);
        fseek(poem, pos_b, SEEK_SET);
        fgets(sent, len + 1, poem);
        cout << sent << endl;
        have_id = 1;
        
    }
    else {
         fread(id_f, sizeof(char), 3, index);
    }
    
  //  fseek(index, 5, SEEK_CUR);
}
if (have_id == 0) {
    cout << "There is no such id" << endl;
}

fclose(index);
fclose(poem);
delete[] str;


return 0;
}
问题排查与修复

原代码存在多处逻辑错误,导致索引结构损坏、读取偏移异常,以下是关键问题点和修复方案:

1. 二进制索引混入文本换行符

构建索引时使用fputs("\n", index)插入换行符,二进制文件不能包含文本格式的换行,这会额外占用字节,导致后续读取时索引指针偏移,无法匹配ID。
修复:删除所有fputs("\n", index)代码。

2. 索引读取分支逻辑错误

当ID不匹配时,原代码用fread(id_f, sizeof(char), 3, index)跳过内容,这完全错误——每个索引条目结构是2字节ID + 4字节起始位置 +4字节长度(共10字节,假设int为4字节),不匹配时应跳过当前条目的起始位置+长度(即2个int,8字节)。
修复:将else分支改为fseek(index, sizeof(int)*2, SEEK_CUR);,直接跳过当前条目的剩余部分。

3. 字符串比较缺少终止符

id_f只读取了2字节ID,但strcmp要求字符串以'\0'结尾,否则会越界读取内存,导致比较结果错误。
修复:每次读取ID后手动添加终止符:id_f[2] = '\0';

4. 句子起始位置计算错误

原代码中pos_b = i +3计算下一个句子的起始位置完全错误,当*pstr是#时,下一个句子的起始位置应为(pstr - str) +1 +2(pstr是#的位置,pstr+1是下一个ID的起始,加2跳过ID)。
修复:重新计算起始位置,确保指向ID后的句子内容起始点。

5. feof循环条件误用

while (!feof(index))会导致最后一次读取失败后仍进入循环,引发错误读取。应使用读取ID的结果作为循环条件,判断是否成功读取到有效ID。
修复:将循环改为while (fread(id_f, sizeof(char), 2, index) == 2)。

修复后的完整代码

#include <iostream>
#include <cstring>
#include <cstdio>

using namespace std;

int main()
{
    FILE *poem, *index;
    const int N = 4096;
    int len;
    char *str = new char[N];

    // 打开原文本文件
    if (!(poem = fopen("poem.txt", "r")))
    {
        cout << "File not found";
        return 0;
    }
    fgets(str, N, poem);

    // 创建二进制索引文件(用.bin后缀明确二进制类型)
    index = fopen("index.bin", "w+b");
    if (!index) {
        cout << "Failed to create index file";
        fclose(poem);
        delete[] str;
        return 0;
    }

    char* pstr = str;
    // 处理第一个条目
    if (*pstr != '\0') {
        // 写入ID(2字节)
        fwrite(pstr, sizeof(char), 2, index);
        // 句子起始位置:跳过2字节ID
        int start_pos = 2;
        fwrite(&start_pos, sizeof(int), 1, index);
        // 找到第一个#,计算句子长度
        char* hash_pos = strchr(pstr, '#');
        if (hash_pos) {
            len = hash_pos - (pstr + 2); // 从ID后到#的长度
            fwrite(&len, sizeof(int), 1, index);
            pstr = hash_pos + 1; // 移动到下一个ID的起始
        }
    }

    // 处理后续条目
    while (*pstr != '\0') {
        // 写入当前ID(2字节)
        fwrite(pstr, sizeof(char), 2, index);
        // 句子起始位置:当前ID的位置 +2(跳过ID)
        int start_pos = (pstr - str) + 2;
        fwrite(&start_pos, sizeof(int), 1, index);
        // 找到下一个#
        char* hash_pos = strchr(pstr, '#');
        if (!hash_pos) break;
        len = hash_pos - (pstr + 2);
        fwrite(&len, sizeof(int), 1, index);
        pstr = hash_pos + 1;
    }

    // 重置文件指针,准备查询
    fseek(index, 0, SEEK_SET);
    fseek(poem, 0, SEEK_SET);

    char id[3] = {0}; // 自动初始化终止符
    char id_f[3] = {0};
    char* sent = new char[N];
    int have_id = 0;

    cout << "Enter the id you want to read: " << endl;
    cin >> id;

    // 循环读取索引条目
    while (fread(id_f, sizeof(char), 2, index) == 2) {
        id_f[2] = '\0'; // 添加字符串终止符
        if (strcmp(id_f, id) == 0) {
            int start_pos, len;
            fread(&start_pos, sizeof(int), 1, index);
            fread(&len, sizeof(int), 1, index);
            // 定位到句子起始位置,读取内容
            fseek(poem, start_pos, SEEK_SET);
            fread(sent, sizeof(char), len, poem);
            sent[len] = '\0'; // 添加终止符
            cout << "Sentence: " << sent << endl;
            have_id = 1;
            break; // 找到后直接退出循环
        } else {
            // 跳过当前条目的起始位置和长度(2个int)
            fseek(index, sizeof(int)*2, SEEK_CUR);
        }
    }

    if (!have_id) {
        cout << "There is no such id" << endl;
    }

    // 释放资源
    fclose(index);
    fclose(poem);
    delete[] str;
    delete[] sent;

    return 0;
}

额外优化点

  • 将索引文件后缀改为.bin,明确是二进制文件,避免和文本文件混淆。
  • 使用strchr替代手动循环找#,代码更简洁可靠。
  • 添加了索引文件打开失败的判断,增强鲁棒性。
  • 找到匹配ID后直接break,无需继续遍历索引。

内容的提问来源于stack exchange,提问作者Kit_ri

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.12 09:05:21