二进制文件导航异常:构建索引后无法按ID读取文本内容
现有文本文件poem.txt,内容为:
10some information#11some more information#12and some more information#
其中10、11、12是固定占2字节的句子ID。需要构建一个二进制索引文件,每条条目包含句子ID(2字节)、句子起始位置(int)、字符串长度(int),用于后续按ID读取对应句子。目前索引文件已生成,但按ID查询时始终提示“不存在该ID”,附上C++代码请求排查。
原代码:
#include <iostream> #include <locale> #include <string.h> using namespace std; int main() { FILE *poem, *index; int N = 4096, len; char *str= new char[N]; if (!(poem = fopen("poem.txt", "r"))) { cout << "File not found"; return 0; } fgets(str, N, poem); index = fopen("index.txt", "w+b"); int pos1 = 2, pos_b; char* pos = str+pos1, *pstr=str; fwrite(pstr, sizeof(char), 2, index); fwrite(&pos1, sizeof(int), 1, index); for (int i = 2; *pstr != '\0'; ++i, pstr++) { if (*pstr == '#') { len = pstr - pos; fwrite(&len, sizeof(int), 1, index); pos = pstr + pos1; if (*(pstr+1)!='\0') { fputs("\n", index); fwrite(pstr+1, sizeof(char), 2, index); pos_b = i + 3; fwrite(&pos_b, sizeof(int), 1, index); } } } pos_b = 0; len = 0; fseek(index, 0, SEEK_SET); fseek(poem, 0, SEEK_SET); char* id = new char[3], * id_f = new char[3], * sent = new char[N]; int have_id = 0; cout << "Enter the id of the offer you want to read" << endl; cin >> id; while (!feof(index)) { fread(id_f, sizeof(char), 2, index); if (strcmp(id_f,id)==0) { fread(&pos_b, sizeof(int), 1, index); fread(&len, sizeof(int), 1, index); fseek(poem, pos_b, SEEK_SET); fgets(sent, len + 1, poem); cout << sent << endl; have_id = 1; } else { fread(id_f, sizeof(char), 3, index); } // fseek(index, 5, SEEK_CUR); } if (have_id == 0) { cout << "There is no such id" << endl; } fclose(index); fclose(poem); delete[] str; return 0; }
原代码存在多处逻辑错误,导致索引结构损坏、读取偏移异常,以下是关键问题点和修复方案:
1. 二进制索引混入文本换行符
构建索引时使用fputs("\n", index)插入换行符,二进制文件不能包含文本格式的换行,这会额外占用字节,导致后续读取时索引指针偏移,无法匹配ID。
修复:删除所有fputs("\n", index)代码。
2. 索引读取分支逻辑错误
当ID不匹配时,原代码用fread(id_f, sizeof(char), 3, index)跳过内容,这完全错误——每个索引条目结构是2字节ID + 4字节起始位置 +4字节长度(共10字节,假设int为4字节),不匹配时应跳过当前条目的起始位置+长度(即2个int,8字节)。
修复:将else分支改为fseek(index, sizeof(int)*2, SEEK_CUR);,直接跳过当前条目的剩余部分。
3. 字符串比较缺少终止符
id_f只读取了2字节ID,但strcmp要求字符串以'\0'结尾,否则会越界读取内存,导致比较结果错误。
修复:每次读取ID后手动添加终止符:id_f[2] = '\0';
4. 句子起始位置计算错误
原代码中pos_b = i +3计算下一个句子的起始位置完全错误,当*pstr是#时,下一个句子的起始位置应为(pstr - str) +1 +2(pstr是#的位置,pstr+1是下一个ID的起始,加2跳过ID)。
修复:重新计算起始位置,确保指向ID后的句子内容起始点。
5. feof循环条件误用
while (!feof(index))会导致最后一次读取失败后仍进入循环,引发错误读取。应使用读取ID的结果作为循环条件,判断是否成功读取到有效ID。
修复:将循环改为while (fread(id_f, sizeof(char), 2, index) == 2)。
修复后的完整代码
#include <iostream> #include <cstring> #include <cstdio> using namespace std; int main() { FILE *poem, *index; const int N = 4096; int len; char *str = new char[N]; // 打开原文本文件 if (!(poem = fopen("poem.txt", "r"))) { cout << "File not found"; return 0; } fgets(str, N, poem); // 创建二进制索引文件(用.bin后缀明确二进制类型) index = fopen("index.bin", "w+b"); if (!index) { cout << "Failed to create index file"; fclose(poem); delete[] str; return 0; } char* pstr = str; // 处理第一个条目 if (*pstr != '\0') { // 写入ID(2字节) fwrite(pstr, sizeof(char), 2, index); // 句子起始位置:跳过2字节ID int start_pos = 2; fwrite(&start_pos, sizeof(int), 1, index); // 找到第一个#,计算句子长度 char* hash_pos = strchr(pstr, '#'); if (hash_pos) { len = hash_pos - (pstr + 2); // 从ID后到#的长度 fwrite(&len, sizeof(int), 1, index); pstr = hash_pos + 1; // 移动到下一个ID的起始 } } // 处理后续条目 while (*pstr != '\0') { // 写入当前ID(2字节) fwrite(pstr, sizeof(char), 2, index); // 句子起始位置:当前ID的位置 +2(跳过ID) int start_pos = (pstr - str) + 2; fwrite(&start_pos, sizeof(int), 1, index); // 找到下一个# char* hash_pos = strchr(pstr, '#'); if (!hash_pos) break; len = hash_pos - (pstr + 2); fwrite(&len, sizeof(int), 1, index); pstr = hash_pos + 1; } // 重置文件指针,准备查询 fseek(index, 0, SEEK_SET); fseek(poem, 0, SEEK_SET); char id[3] = {0}; // 自动初始化终止符 char id_f[3] = {0}; char* sent = new char[N]; int have_id = 0; cout << "Enter the id you want to read: " << endl; cin >> id; // 循环读取索引条目 while (fread(id_f, sizeof(char), 2, index) == 2) { id_f[2] = '\0'; // 添加字符串终止符 if (strcmp(id_f, id) == 0) { int start_pos, len; fread(&start_pos, sizeof(int), 1, index); fread(&len, sizeof(int), 1, index); // 定位到句子起始位置,读取内容 fseek(poem, start_pos, SEEK_SET); fread(sent, sizeof(char), len, poem); sent[len] = '\0'; // 添加终止符 cout << "Sentence: " << sent << endl; have_id = 1; break; // 找到后直接退出循环 } else { // 跳过当前条目的起始位置和长度(2个int) fseek(index, sizeof(int)*2, SEEK_CUR); } } if (!have_id) { cout << "There is no such id" << endl; } // 释放资源 fclose(index); fclose(poem); delete[] str; delete[] sent; return 0; }
额外优化点
- 将索引文件后缀改为
.bin,明确是二进制文件,避免和文本文件混淆。 - 使用
strchr替代手动循环找#,代码更简洁可靠。 - 添加了索引文件打开失败的判断,增强鲁棒性。
- 找到匹配ID后直接
break,无需继续遍历索引。
内容的提问来源于stack exchange,提问作者Kit_ri

