You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用C语言实现支持命令行参数的双模式文件检索工具

基于C语言的文件检索工具完善需求

功能需求

  • 数据文件为两列CSV格式,第一列是待查找数字,第二列是需展示的关联数据;
  • 工具通过命令行参数接收指令,格式为find [-s | -t] number:
    • number:待查找的目标数字;
    • -s:采用逐行扫描文件的方式查找;
    • -t:采用自定义哈希方案查找;
  • 查找成功时输出:Found [number] with data on second column is [关联数据],并显示查找耗时;
  • 参数错误时输出提示:
    No method defined
    Proper Syntax is
    find [ -s | -t ] number
    

当前代码存在的问题

现有C代码存在以下缺陷,需要修改完善:

  • 硬编码查找值,未通过命令行参数接收目标数字和查找方式;
  • 未读取CSV文件的第二列关联数据;
  • 无查找耗时统计功能;
  • 哈希查找逻辑错误,未正确实现哈希表存储与查找逻辑;
  • 文件读取依赖feof判断循环结束,存在逻辑风险。

完善后的代码实现

#include <stdio.h>
#include <stdlib.h>
#include <stdbool.h>
#include <time.h>
#include <string.h>

#define HASH_SIZE 100
#define MAX_DATA_LEN 256

// 哈希表节点结构
typedef struct HashNode {
    int key;
    char data[MAX_DATA_LEN];
    struct HashNode* next;
} HashNode;

HashNode* hash_table[HASH_SIZE] = {NULL};

// 哈希函数(处理负数确保非负索引)
int hash(int value) {
    int res = value % HASH_SIZE;
    return res < 0 ? res + HASH_SIZE : res;
}

// 向哈希表插入节点
void hash_insert(int key, const char* data) {
    int index = hash(key);
    HashNode* new_node = (HashNode*)malloc(sizeof(HashNode));
    if (!new_node) {
        perror("Failed to allocate memory");
        return;
    }
    new_node->key = key;
    strncpy(new_node->data, data, MAX_DATA_LEN - 1);
    new_node->data[MAX_DATA_LEN - 1] = '\0';
    new_node->next = hash_table[index];
    hash_table[index] = new_node;
}

// 从哈希表查找节点
HashNode* hash_find(int key) {
    int index = hash(key);
    HashNode* current = hash_table[index];
    while (current) {
        if (current->key == key) {
            return current;
        }
        current = current->next;
    }
    return NULL;
}

// 释放哈希表内存
void hash_free() {
    for (int i = 0; i < HASH_SIZE; i++) {
        HashNode* current = hash_table[i];
        while (current) {
            HashNode* temp = current;
            current = current->next;
            free(temp);
        }
        hash_table[i] = NULL;
    }
}

// 逐行扫描查找
bool scan_find(FILE* file, int target, char* result_data) {
    rewind(file); // 重置文件指针到开头
    int current_key;
    char current_data[MAX_DATA_LEN];
    while (fscanf(file, "%d,%s", &current_key, current_data) == 2) {
        if (current_key == target) {
            strncpy(result_data, current_data, MAX_DATA_LEN - 1);
            result_data[MAX_DATA_LEN - 1] = '\0';
            return true;
        }
    }
    return false;
}

// 哈希方式查找(先构建哈希表再查找)
bool hash_table_find(FILE* file, int target, char* result_data) {
    rewind(file);
    // 构建哈希表
    int current_key;
    char current_data[MAX_DATA_LEN];
    while (fscanf(file, "%d,%s", &current_key, current_data) == 2) {
        hash_insert(current_key, current_data);
    }
    // 查找目标
    HashNode* node = hash_find(target);
    if (node) {
        strncpy(result_data, node->data, MAX_DATA_LEN - 1);
        result_data[MAX_DATA_LEN - 1] = '\0';
        return true;
    }
    return false;
}

int main(int argc, char* argv[]) {
    // 校验命令行参数格式
    if (argc != 3 || (strcmp(argv[1], "-s") != 0 && strcmp(argv[1], "-t") != 0)) {
        printf("No method defined\nProper Syntax is\nfind [ -s | -t ] number\n");
        return 1;
    }

    int target = atoi(argv[2]);
    FILE* file = fopen("DATA_FILE.csv", "r");
    if (!file) {
        perror("Error while opening the file");
        return 1;
    }

    char result_data[MAX_DATA_LEN];
    bool found = false;
    clock_t start, end;
    double elapsed_time;

    // 记录查找开始时间
    start = clock();
    if (strcmp(argv[1], "-s") == 0) {
        found = scan_find(file, target, result_data);
    } else if (strcmp(argv[1], "-t") == 0) {
        found = hash_table_find(file, target, result_data);
    }
    // 记录查找结束时间并计算耗时
    end = clock();
    elapsed_time = ((double)(end - start)) / CLOCKS_PER_SEC;

    // 输出结果
    if (found) {
        printf("Found %d with data on second column is %s\n", target, result_data);
        printf("Search took %.6f seconds\n", elapsed_time);
    } else {
        printf("Value %d not found in file\n", target);
    }

    // 清理资源
    if (strcmp(argv[1], "-t") == 0) {
        hash_free();
    }
    fclose(file);
    return 0;
}

代码改进说明

  1. 命令行参数处理:严格校验参数数量与格式,不符合要求时输出规范错误提示;
  2. 双查找模式实现:
    • -s模式:逐行扫描CSV文件,匹配目标数字后返回关联数据;
    • -t模式:先将文件内容加载到链式哈希表(解决哈希冲突),再通过哈希表快速查找;
  3. 耗时统计:使用clock()函数记录查找前后的时间差,计算并输出精确到微秒的耗时;
  4. CSV解析优化:使用fscanf("%d,%s")正确解析两列数据,支持字符串类型的关联数据;
  5. 内存管理:哈希表使用完成后主动释放内存,避免内存泄漏;
  6. 错误处理:添加文件打开失败、内存分配失败的错误提示,提升程序健壮性。

内容的提问来源于stack exchange,提问作者randomized

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.09 15:01:05