You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在C语言中存储字符串并统计文本文件内单词出现次数

需求说明

我有一个包含以下内容的文本文件:

we are studying hard
we can go out for vacation
I can have fun!!
I can have fun!!

fun!

期望输出为:

#1  we       2
#2  are      1
#3  studying 1
#4  hard     1
#5  can      3
#6  go       1
#7  out      1
#8  for      1
#9  vacation 1
#10 I        2
#11 have     2
#12 fun      3

需要统计每个单词的重复出现次数。我尝试编写了以下代码,但仅能输出内容,无法实现计数功能,请求帮忙修正代码以实现需求:

#include <stdio.h>
#include <stdlib.h>
    
#define MAX_LEN 1000
    
int main(int argc, char *argv[]) {
    char data[MAX_LEN];

    if (argc < 2) {
        printf("Usage: %s filename\n", argv[0]);
        return 1;
    }
    char *filename = argv[1];
    FILE *file = fopen(filename, "r");

    if (file == NULL) {
        perror("Error opening file");
        return(1);
    }
    int line = 0;

    char c;
    while ((c = fgetc(file)) != EOF)
    {
        if (c == ' ' || c == '\n')
        {
            printf("\n");
        }
        else
        {
            line++;
            printf("%c", c);
        }
    }

    for (int i = 0; i < line; i++)
        printf("%s", data[i]);

    fclose(file);
    return 0;
}
修正后的代码实现

要实现单词计数,需要完成提取单词(过滤标点)、存储单词并统计次数、格式化输出这几个核心步骤。以下是修正后的完整代码:

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <ctype.h>

#define MAX_WORDS 100   // 最多存储的不同单词数量
#define MAX_WORD_LEN 50 // 单个单词的最大长度

// 存储单词和对应计数的结构体
typedef struct {
    char word[MAX_WORD_LEN];
    int count;
} WordCount;

// 判断字符是否为字母(用于过滤标点符号)
int is_letter(char c) {
    return isalpha((unsigned char)c);
}

int main(int argc, char *argv[]) {
    WordCount words[MAX_WORDS] = {0}; // 初始化单词数组
    int unique_word_count = 0;        // 已存储的不同单词数量

    // 检查命令行参数
    if (argc < 2) {
        printf("Usage: %s filename\n", argv[0]);
        return 1;
    }

    // 打开文件
    FILE *file = fopen(argv[1], "r");
    if (file == NULL) {
        perror("Error opening file");
        return 1;
    }

    char current_word[MAX_WORD_LEN];
    int current_word_index = 0;
    char c;

    // 逐字符读取文件,提取单词
    while ((c = fgetc(file)) != EOF) {
        if (is_letter(c)) {
            // 将字母转为小写,统一统计(避免大小写差异导致重复计数)
            current_word[current_word_index++] = tolower((unsigned char)c);
            // 防止单词长度超过限制
            if (current_word_index >= MAX_WORD_LEN - 1) {
                current_word_index = MAX_WORD_LEN - 1;
            }
        } else {
            // 遇到非字母字符,结束当前单词的提取
            if (current_word_index > 0) {
                current_word[current_word_index] = '\0'; // 添加字符串结束符
                current_word_index = 0;

                // 检查单词是否已存在
                int found = 0;
                for (int i = 0; i < unique_word_count; i++) {
                    if (strcmp(words[i].word, current_word) == 0) {
                        words[i].count++;
                        found = 1;
                        break;
                    }
                }

                // 单词不存在则添加到数组
                if (!found && unique_word_count < MAX_WORDS) {
                    strcpy(words[unique_word_count].word, current_word);
                    words[unique_word_count].count = 1;
                    unique_word_count++;
                }
            }
        }
    }

    // 处理文件末尾可能剩余的未完成单词
    if (current_word_index > 0) {
        current_word[current_word_index] = '\0';
        int found = 0;
        for (int i = 0; i < unique_word_count; i++) {
            if (strcmp(words[i].word, current_word) == 0) {
                words[i].count++;
                found = 1;
                break;
            }
        }
        if (!found && unique_word_count < MAX_WORDS) {
            strcpy(words[unique_word_count].word, current_word);
            words[unique_word_count].count = 1;
            unique_word_count++;
        }
    }

    // 按要求格式化输出结果
    for (int i = 0; i < unique_word_count; i++) {
        printf("#%d  %-8s %d\n", i+1, words[i].word, words[i].count);
    }

    fclose(file);
    return 0;
}
代码关键说明
  1. 结构体管理:用WordCount结构体将单词和计数绑定,方便统一存储和遍历。
  2. 单词提取与标准化:只保留字母字符,并统一转为小写,确保"We"和"we"被算作同一个单词。
  3. 计数逻辑:每提取一个单词,遍历已存储的单词列表,存在则计数加1,不存在则新增条目。
  4. 格式对齐:用%-8s让单词占固定8个字符宽度,和期望输出格式保持一致。
  5. 边界处理:加入单词长度和数组容量限制,避免溢出;同时处理文件末尾未完成的单词提取。

内容的提问来源于stack exchange,提问作者klp

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.01 18:01:23