You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言中如何捕获输入文件换行符并读取后续单词

问题:读取文件时无法捕获换行符后的单词

需求:打开文件,计算平均单词长度,输出所有长度大于平均值的单词。目前代码无法捕获换行符后的单词(如示例中标记为*XXX<*的单词),但逗号、句号后的单词可正常读取。

示例文件内容:

Dijkstra,
*CACM,<*
*15:10,<*
people
belittle
ambitions.
*Small,<*
people
*always,<*
that,
really
great
become
great.

原代码:

#include <stdio.h>
#include <stdlib.h>
#include <ctype.h>

#define MAX_WORD_LENGTH 30
#define MAX_FILE_SIZE 2048

int main(int argc, char *argv[]) {
    if (argc != 2) {
        printf("Usage: %s <filename>\n", argv[0]);
        return 1;
    }

    // Open the file
    FILE *file = fopen(argv[1], "r");
    if (file == NULL) {
        perror("Error opening file");
        return 1;
    }

    int total_length = 0;
    int word_count = 0;
    int max_word_length = 0;
    char word[MAX_WORD_LENGTH];
    while (fscanf(file, "%s", word) == 1) {
        int length = 0;
        for (int i = 0; word[i] != '\0'; i++) {
            if (isalnum(word[i])) {
                length++;
            }
        }
        if (length > 0) { 
            total_length += length;
            word_count++;
            if (length > max_word_length) {
                max_word_length = length;
            }
        }
    }


    int average_length = 0;
    if (word_count > 0) {
        average_length = total_length / word_count;
    }


    rewind(file);

    while (fscanf(file, "%s", word) == 1) {
        int length = 0;
        for (int i = 0; word[i] != '\0'; i++) {
            if (isalnum(word[i])) { // Check if character is alphanumeric
                length++;
            }
        }
        if (length > average_length) {
            printf("%s\n", word);
        }
    }
    //printf("\n");

    fclose(file);

    return 0;
}

问题分析与解决方法

核心问题澄清

你的代码里fscanf("%s", word)可以正常读取换行后的单词(包括*CACM,<*这类格式的单词),它们没出现在输出中的原因是:这些单词的有效长度(仅统计字母数字字符的数量)没有超过整数除法计算出的平均长度。

以示例文件为例,整数除法得到的平均长度是5,而:

  • *CACM,<*的有效长度是4(仅CACM)
  • *Small,<*的有效长度是5(仅Small)
  • *always,<*的有效长度是5(仅always)

这些都不满足“长度大于平均值”的条件,因此不会被输出。

优化方案

1. 用浮点数计算平均长度,避免截断误差

将整数除法改为浮点数除法,能更精准判断单词长度是否大于平均值,避免有效长度接近平均值时被误判。

修改后的代码:

#include <stdio.h>
#include <stdlib.h>
#include <ctype.h>

#define MAX_WORD_LENGTH 30

int main(int argc, char *argv[]) {
    if (argc != 2) {
        printf("Usage: %s <filename>\n", argv[0]);
        return 1;
    }

    FILE *file = fopen(argv[1], "r");
    if (file == NULL) {
        perror("Error opening file");
        return 1;
    }

    int total_length = 0;
    int word_count = 0;
    char word[MAX_WORD_LENGTH];

    // 第一遍遍历:统计总有效长度和单词数
    while (fscanf(file, "%s", word) == 1) {
        int length = 0;
        for (int i = 0; word[i] != '\0'; i++) {
            // 转换为unsigned char避免isalnum的未定义行为
            if (isalnum((unsigned char)word[i])) {
                length++;
            }
        }
        if (length > 0) { 
            total_length += length;
            word_count++;
        }
    }

    double average_length = 0.0;
    if (word_count > 0) {
        average_length = (double)total_length / word_count;
        printf("Average word length: %.2f\n", average_length);
    }

    rewind(file);

    // 第二遍遍历:输出有效长度大于平均值的单词
    while (fscanf(file, "%s", word) == 1) {
        int length = 0;
        for (int i = 0; word[i] != '\0'; i++) {
            if (isalnum((unsigned char)word[i])) {
                length++;
            }
        }
        if (length > average_length) {
            printf("%s\n", word);
        }
    }

    fclose(file);
    return 0;
}

2. 自定义单词分割逻辑(逐字符读取)

如果希望将“单词”定义为纯字母数字序列(而非包含*、<等符号的字符串),可以改用逐字符读取的方式,精准提取单词:

#include <stdio.h>
#include <stdlib.h>
#include <ctype.h>

#define MAX_WORD_LENGTH 30

int main(int argc, char *argv[]) {
    if (argc != 2) {
        printf("Usage: %s <filename>\n", argv[0]);
        return 1;
    }

    FILE *file = fopen(argv[1], "r");
    if (file == NULL) {
        perror("Error opening file");
        return 1;
    }

    int total_length = 0;
    int word_count = 0;
    char word[MAX_WORD_LENGTH];
    int pos = 0;
    int c;

    // 第一遍:逐字符提取纯字母数字单词并统计
    while ((c = fgetc(file)) != EOF) {
        if (isalnum((unsigned char)c)) {
            if (pos < MAX_WORD_LENGTH - 1) {
                word[pos++] = c;
            }
        } else {
            if (pos > 0) {
                word[pos] = '\0';
                total_length += pos;
                word_count++;
                pos = 0;
            }
        }
    }
    // 处理文件末尾的单词
    if (pos > 0) {
        word[pos] = '\0';
        total_length += pos;
        word_count++;
    }

    double average_length = 0.0;
    if (word_count > 0) {
        average_length = (double)total_length / word_count;
        printf("Average word length: %.2f\n", average_length);
    }

    rewind(file);
    pos = 0;

    // 第二遍:输出长度大于平均值的纯字母数字单词
    while ((c = fgetc(file)) != EOF) {
        if (isalnum((unsigned char)c)) {
            if (pos < MAX_WORD_LENGTH - 1) {
                word[pos++] = c;
            }
        } else {
            if (pos > 0) {
                word[pos] = '\0';
                if (pos > average_length) {
                    printf("%s\n", word);
                }
                pos = 0;
            }
        }
    }
    // 处理末尾单词
    if (pos > 0) {
        word[pos] = '\0';
        if (pos > average_length) {
            printf("%s\n", word);
        }
    }

    fclose(file);
    return 0;
}

这个版本会将*CACM,<*中的CACM提取为独立单词,统计和输出的都是纯字母数字内容,更符合常规的“单词”定义。


内容的提问来源于stack exchange,提问作者user23760512

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.26 20:00:55