如何在C语言中存储字符串并统计文本文件内单词出现次数
需求说明
我有一个包含以下内容的文本文件:
we are studying hard we can go out for vacation I can have fun!! I can have fun!! fun!
期望输出为:
#1 we 2 #2 are 1 #3 studying 1 #4 hard 1 #5 can 3 #6 go 1 #7 out 1 #8 for 1 #9 vacation 1 #10 I 2 #11 have 2 #12 fun 3
需要统计每个单词的重复出现次数。我尝试编写了以下代码,但仅能输出内容,无法实现计数功能,请求帮忙修正代码以实现需求:
#include <stdio.h> #include <stdlib.h> #define MAX_LEN 1000 int main(int argc, char *argv[]) { char data[MAX_LEN]; if (argc < 2) { printf("Usage: %s filename\n", argv[0]); return 1; } char *filename = argv[1]; FILE *file = fopen(filename, "r"); if (file == NULL) { perror("Error opening file"); return(1); } int line = 0; char c; while ((c = fgetc(file)) != EOF) { if (c == ' ' || c == '\n') { printf("\n"); } else { line++; printf("%c", c); } } for (int i = 0; i < line; i++) printf("%s", data[i]); fclose(file); return 0; }
修正后的代码实现
要实现单词计数,需要完成提取单词(过滤标点)、存储单词并统计次数、格式化输出这几个核心步骤。以下是修正后的完整代码:
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <ctype.h> #define MAX_WORDS 100 // 最多存储的不同单词数量 #define MAX_WORD_LEN 50 // 单个单词的最大长度 // 存储单词和对应计数的结构体 typedef struct { char word[MAX_WORD_LEN]; int count; } WordCount; // 判断字符是否为字母(用于过滤标点符号) int is_letter(char c) { return isalpha((unsigned char)c); } int main(int argc, char *argv[]) { WordCount words[MAX_WORDS] = {0}; // 初始化单词数组 int unique_word_count = 0; // 已存储的不同单词数量 // 检查命令行参数 if (argc < 2) { printf("Usage: %s filename\n", argv[0]); return 1; } // 打开文件 FILE *file = fopen(argv[1], "r"); if (file == NULL) { perror("Error opening file"); return 1; } char current_word[MAX_WORD_LEN]; int current_word_index = 0; char c; // 逐字符读取文件,提取单词 while ((c = fgetc(file)) != EOF) { if (is_letter(c)) { // 将字母转为小写,统一统计(避免大小写差异导致重复计数) current_word[current_word_index++] = tolower((unsigned char)c); // 防止单词长度超过限制 if (current_word_index >= MAX_WORD_LEN - 1) { current_word_index = MAX_WORD_LEN - 1; } } else { // 遇到非字母字符,结束当前单词的提取 if (current_word_index > 0) { current_word[current_word_index] = '\0'; // 添加字符串结束符 current_word_index = 0; // 检查单词是否已存在 int found = 0; for (int i = 0; i < unique_word_count; i++) { if (strcmp(words[i].word, current_word) == 0) { words[i].count++; found = 1; break; } } // 单词不存在则添加到数组 if (!found && unique_word_count < MAX_WORDS) { strcpy(words[unique_word_count].word, current_word); words[unique_word_count].count = 1; unique_word_count++; } } } } // 处理文件末尾可能剩余的未完成单词 if (current_word_index > 0) { current_word[current_word_index] = '\0'; int found = 0; for (int i = 0; i < unique_word_count; i++) { if (strcmp(words[i].word, current_word) == 0) { words[i].count++; found = 1; break; } } if (!found && unique_word_count < MAX_WORDS) { strcpy(words[unique_word_count].word, current_word); words[unique_word_count].count = 1; unique_word_count++; } } // 按要求格式化输出结果 for (int i = 0; i < unique_word_count; i++) { printf("#%d %-8s %d\n", i+1, words[i].word, words[i].count); } fclose(file); return 0; }
代码关键说明
- 结构体管理:用
WordCount结构体将单词和计数绑定,方便统一存储和遍历。 - 单词提取与标准化:只保留字母字符,并统一转为小写,确保"We"和"we"被算作同一个单词。
- 计数逻辑:每提取一个单词,遍历已存储的单词列表,存在则计数加1,不存在则新增条目。
- 格式对齐:用
%-8s让单词占固定8个字符宽度,和期望输出格式保持一致。 - 边界处理:加入单词长度和数组容量限制,避免溢出;同时处理文件末尾未完成的单词提取。
内容的提问来源于stack exchange,提问作者klp
相关产品推荐
相关产品推荐

