You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用strtok_r()解析文本文件 匹配序列号提取对应行数据

问题根因

你二次调用strtok拆分字段出现异常,是两个逻辑错误导致的:

  • 二维数组matrice的容量是估算值:固定单行长度25字节、按文件大小反推行数的逻辑非常容易触发缓冲区溢出,只要存在单行长度超过25字节、或者实际行数大于估算值的情况,就会发生内存越界,直接导致程序运行异常。
  • 字符串拆分上下文混用:strtok_r拆分完所有行之后,行拆分的上下文指针svptr1已经指向原缓冲区的末尾位置,如果你直接复用这个上下文去拆分行内字段,必然会访问非法内存地址。

另外你一次性把整个文件读入内存再拆分的逻辑完全没必要,还额外引入了VLA(可变长度数组)的兼容问题,逐行读取处理的逻辑更简单可靠。

修正实现

直接逐行读取文件,每读一行就拆分首字段做匹配,不需要提前估算数组大小,也不会有内存越界风险,完整可运行代码如下:

#include <stdio.h>
#include <string.h>
#include <stdlib.h>

// 定义单条记录的结构,匹配文件4个字段的固定格式
#define MAX_FIELD_LEN 64
#define MAX_MATCH_RECORDS 1024
typedef struct {
    char serial[MAX_FIELD_LEN];
    char code[MAX_FIELD_LEN];
    char num[MAX_FIELD_LEN];
    char value[MAX_FIELD_LEN];
} Record;

int main()
{
    char target_serial[50];  
    printf("insert serial number: \n");
    // 限制scanf读取长度,避免输入过长溢出
    scanf("%49s", target_serial);
    // 清掉输入缓冲区残留的换行符,避免后续fgets读空行
    while(getchar() != '\n');

    FILE *fp = fopen("prova.txt", "r");
    if (!fp){
        printf("file doesnt exist\n");    
        return -1;        
    }

    Record match_records[MAX_MATCH_RECORDS];
    int match_count = 0;
    char line_buf[1024]; // 行缓冲区设1024字节,足够覆盖常规文本行
    char *field_ctx;
    const char *field_delim = ";";

    // 逐行读取文件
    while (fgets(line_buf, sizeof(line_buf), fp) != NULL) {
        // 先拆分首字段,和目标序列号比对
        char *first_field = strtok_r(line_buf, field_delim, &field_ctx);
        if (first_field == NULL) continue; // 跳过空行

        if (strcmp(first_field, target_serial) == 0) {
            // 序列号匹配,读取剩余3个字段
            if (match_count >= MAX_MATCH_RECORDS) {
                printf("match records exceed max limit\n");
                break;
            }
            Record *cur = &match_records[match_count];
            strncpy(cur->serial, first_field, MAX_FIELD_LEN - 1);
            cur->serial[MAX_FIELD_LEN - 1] = '\0';

            char *field = strtok_r(NULL, field_delim, &field_ctx);
            if (field) {
                strncpy(cur->code, field, MAX_FIELD_LEN - 1);
                cur->code[MAX_FIELD_LEN - 1] = '\0';
            }
            field = strtok_r(NULL, field_delim, &field_ctx);
            if (field) {
                strncpy(cur->num, field, MAX_FIELD_LEN - 1);
                cur->num[MAX_FIELD_LEN - 1] = '\0';
            }
            field = strtok_r(NULL, field_delim, &field_ctx);
            if (field) {
                // 手动去掉字段末尾的换行符,兼容Windows/Unix换行格式
                char *newline = strchr(field, '\n');
                if (newline) *newline = '\0';
                char *cr = strchr(field, '\r');
                if (cr) *cr = '\0';
                strncpy(cur->value, field, MAX_FIELD_LEN - 1);
                cur->value[MAX_FIELD_LEN - 1] = '\0';
            }
            match_count++;
        }
    }

    fclose(fp);

    // 测试输出匹配到的记录
    printf("found %d match records:\n", match_count);
    for (int i = 0; i < match_count; i++) {
        printf("record %d: %s | %s | %s | %s\n", 
            i+1, 
            match_records[i].serial,
            match_records[i].code,
            match_records[i].num,
            match_records[i].value
        );
    }

    return 0;
}

关键注意点
  • 所有字符串拷贝操作都加了长度限制,避免缓冲区溢出,比无长度限制的strcpy安全很多。
  • 每一行的字段拆分都用独立的上下文指针field_ctx,不会和其他逻辑的拆分状态冲突,从根源上避免strtok系列函数的状态混用问题。
  • 逐行读取的内存占用固定为1KB左右,哪怕文件大小到几百MB也不会出现内存不足的问题,比一次性读入整个文件的方案兼容性更好。

内容的提问来源于stack exchange,提问作者al366io

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.29 00:27:27