You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何修复C语言中从分号分隔CSV填充两个数组的代码?

C语言CSV解析代码修复方案

问题说明

需要修复一段C语言代码,实现从分号分隔的CSV文件中读取内容,分别填充到两个独立数组中。CSV内容如下:

to meet/encounter;begegnen
to care for/look after;betreuen
superior;vorgesetzt
to borrow/lend/rent;ausleihen

现有代码运行后输出混乱,出现内容截断、错位的情况,比如:

English: to meet
German: ow/lend
English:  encou
German: leihen
English: nter
German:
English: gegnen

错误分析

  1. fgets读取长度错误:fgets(line, sizeof(line), fptr)中,line是指针类型,sizeof(line)返回的是指针本身的大小(通常4或8字节),而非分配的LINE_SIZE,导致每次只读取少量字符,行内容被截断,后续分割自然混乱。
  2. 重复调用strtok:代码中连续两次调用token = strtok(line, ";"),第二次调用会重新开始分割当前行,属于冗余操作。
  3. 二维数组内存分配不完整:全局变量source_vocab和target_vocab是二级指针,但现有代码只给每个元素分配了内存,未先为二级指针本身分配足够的空间(即存储多个char*指针的数组)。

修复后的完整代码

#include <stdio.h>
#include <stdlib.h>
#include <string.h>

#define LINE_SIZE 100
#define STRING_SIZE 55
#define RED "\033[0;31m"

char **source_vocab = NULL;
char **target_vocab = NULL;

// 为二级指针分配外层数组内存,再为每个元素分配字符串内存
void allocate_mem(char ***dictionary, int n)
{
    // 先分配存储指针的数组
    *dictionary = malloc(n * sizeof(char*));
    if (*dictionary == NULL)
    {
        printf("%sMemory Error. Exiting..\n", RED);
        exit(EXIT_FAILURE);
    }
    // 为每个指针分配字符串内存
    for(int i = 0; i < n; i++)
    {
        (*dictionary)[i] = malloc(STRING_SIZE * sizeof(char));
        if ((*dictionary)[i] == NULL)
        {
            printf("%sMemory Error. Exiting..\n", RED);
            exit(EXIT_FAILURE);
        }
    }
}

// 统计CSV文件的行数,用于提前分配内存
int count_lines(FILE *fptr)
{
    int count = 0;
    char line[LINE_SIZE];
    while(fgets(line, LINE_SIZE, fptr) != NULL)
    {
        // 跳过空行
        if (strlen(line) > 1)
            count++;
    }
    // 文件指针重置到开头
    fseek(fptr, 0, SEEK_SET);
    return count;
}

void populate_arrs(FILE *fptr)
{
    char *line = malloc(LINE_SIZE * sizeof(char));
    if (line == NULL)
    {
        printf("%sMemory Error. Exiting..\n", RED);
        exit(EXIT_FAILURE);
    }
    int count = 0;
    int total_lines = count_lines(fptr);

    // 为两个词汇数组分配内存
    allocate_mem(&source_vocab, total_lines);
    allocate_mem(&target_vocab, total_lines);

    while(fgets(line, LINE_SIZE, fptr) != NULL)
    {
        // 去掉换行符
        line[strcspn(line, "\n")] = '\0';
        // 跳过空行
        if (strlen(line) == 0)
            continue;

        // 第一次调用strtok初始化分割
        char *token = strtok(line, ";");
        
        if(token != NULL)
            strncpy(source_vocab[count], token, STRING_SIZE - 1);
        // 确保字符串以'\0'结尾
        source_vocab[count][STRING_SIZE - 1] = '\0';

        token = strtok(NULL, ";");
        if(token != NULL)
            strncpy(target_vocab[count], token, STRING_SIZE - 1);
        target_vocab[count][STRING_SIZE - 1] = '\0';

        count++;
    }
    free(line);
}

// 测试用例
int main()
{
    FILE *fptr = fopen("vocab.csv", "r");
    if (fptr == NULL)
    {
        printf("%sFailed to open file.\n", RED);
        exit(EXIT_FAILURE);
    }

    populate_arrs(fptr);
    fclose(fptr);

    // 打印结果
    for(int i = 0; i < 4; i++)
    {
        printf("English: %s\n", source_vocab[i]);
        printf("German: %s\n\n", target_vocab[i]);
    }

    // 释放内存
    for(int i = 0; i < 4; i++)
    {
        free(source_vocab[i]);
        free(target_vocab[i]);
    }
    free(source_vocab);
    free(target_vocab);

    return 0;
}

关键修复点说明

  1. 修正fgets读取长度:将sizeof(line)改为LINE_SIZE,确保每次读取完整的一行内容。
  2. 完善内存分配逻辑:新增count_lines函数统计文件行数,先为二级指针分配存储指针的数组空间,再为每个元素分配字符串内存,避免数组越界。
  3. 替换strdup为strncpy:strdup会自动分配内存,这里使用strncpy并手动添加结束符,更符合预先分配的逻辑,避免内存泄漏。
  4. 移除冗余的strtok调用:只保留第一次strtok(line, ";")初始化分割,后续用strtok(NULL, ";")获取下一个token。
  5. 添加空行处理:跳过文件中的空行,避免无效数据填充到数组中。

内容的提问来源于stack exchange,提问作者Daniel

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.20 13:03:14