You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言指定单词文件词频统计:仅首个单词统计有效问题求助

单词频率统计代码修复方案

需求:从指定文件读取待统计单词列表,在另一文本文件中统计这些单词的出现频率。

待统计单词文件(file)内容:

3
watch
words
become

目标文本文件(file1)内容:

watch become words
watch words become
watch words watch
become
watch
become

原实现C语言代码:

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define N 100

typedef struct {
    char word[N+1];
    int occurrences;
} index_t;

int main()
{
    int i=0,j=0;
    FILE *file;
    FILE *file1;
    file=fopen("file","r");
    file1=fopen("file1","r");

    if(file1==NULL || file==NULL)
    {
        printf("Error in the file");
        exit(1);
    }
    int n;
    fscanf(file,"%d",&n);

    index_t v[n];
    char str[N];
    for(i=0;i<n;i++)
    {
        fscanf(file,"%s",v[i].word);
        v[i].occurrences=0;
        while(fscanf(file1,"%s",str)!=EOF)
        {
            if(strcmp(v[i].word,str)==0)
            {
                v[i].occurrences++;
            }
        }
        printf("%s:%d occurrences\n",v[i].word,v[i].occurrences);
    }

    return 0;
}

问题原因

原代码的核心问题是:统计第一个单词时,while(fscanf(file1,"%s",str)!=EOF)循环已经把file1的文件指针移动到了文件末尾(EOF位置)。后续循环统计其他单词时,fscanf直接返回EOF,不会再读取任何内容,导致后续单词的出现次数始终为0。

修复方案

下面提供两种可行的修复思路:

方案1:每次统计前重置文件指针到开头

在每次进入单个单词的统计循环前,调用rewind(file1)将file1的文件指针重置到文件起始位置,确保每次统计都能重新读取整个目标文件内容。

修复后的代码:

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define N 100

typedef struct {
    char word[N+1];
    int occurrences;
} index_t;

int main()
{
    int i=0;
    FILE *file;
    FILE *file1;
    file=fopen("file","r");
    file1=fopen("file1","r");

    if(file1==NULL || file==NULL)
    {
        printf("Error in the file");
        exit(1);
    }
    int n;
    fscanf(file,"%d",&n);

    index_t v[n];
    char str[N];
    for(i=0;i<n;i++)
    {
        fscanf(file,"%s",v[i].word);
        v[i].occurrences=0;
        // 重置文件指针到file1开头
        rewind(file1);
        while(fscanf(file1,"%s",str)!=EOF)
        {
            if(strcmp(v[i].word,str)==0)
            {
                v[i].occurrences++;
            }
        }
        printf("%s:%d occurrences\n",v[i].word,v[i].occurrences);
    }

    // 关闭文件,避免资源泄漏
    fclose(file);
    fclose(file1);
    return 0;
}

方案2:先读取目标文件所有单词到内存

如果目标文件体积较大,多次重复读取会影响效率,可以先把file1中的所有单词读取到内存数组中,之后直接在内存中遍历匹配每个待统计单词。这种方法只需要读取一次目标文件,适合大文件场景。

示例代码:

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define N 100
#define MAX_WORDS 1000 // 根据实际文件大小调整

typedef struct {
    char word[N+1];
    int occurrences;
} index_t;

int main()
{
    int i=0, word_count=0;
    FILE *file;
    FILE *file1;
    file=fopen("file","r");
    file1=fopen("file1","r");

    if(file1==NULL || file==NULL)
    {
        printf("Error in the file");
        exit(1);
    }

    // 先读取目标文件所有单词到内存数组
    char all_words[MAX_WORDS][N+1];
    while(fscanf(file1,"%s",all_words[word_count])!=EOF && word_count < MAX_WORDS)
    {
        word_count++;
    }

    int n;
    fscanf(file,"%d",&n);
    index_t v[n];

    for(i=0;i<n;i++)
    {
        fscanf(file,"%s",v[i].word);
        v[i].occurrences=0;
        // 遍历内存中的单词数组统计次数
        for(int j=0;j<word_count;j++)
        {
            if(strcmp(v[i].word, all_words[j])==0)
            {
                v[i].occurrences++;
            }
        }
        printf("%s:%d occurrences\n",v[i].word,v[i].occurrences);
    }

    fclose(file);
    fclose(file1);
    return 0;
}

验证结果

修复后运行代码,会输出正确的统计结果:

watch:4 occurrences
words:3 occurrences
become:4 occurrences

内容的提问来源于stack exchange,提问作者Severjan Lici

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.27 16:07:42