You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言printf输出出现意外空格,求代码异常原因解析

行数/单词数/字节数统计函数的调试异常分析

问题背景

我编写了一个C语言counter函数用于统计文本文件的行数、单词数和字节数,但调试时出现异常:输出中T h e前出现了不属于当前行的4个空格,且该内容本该属于下一次循环的输出。我怀疑这和文件第二行的4个空格有关,但不清楚错位原因。

实现代码

#include <stdio.h>
#include <stdlib.h>
#include <stdbool.h>
#include <ctype.h>
#include <limits.h> // 包含LINE_MAX定义

int *counter(const char *filename, bool bytes_flag, bool words_flag, bool lines_flag) {
    int *counts = malloc(sizeof(int) * 3);
    int line_count = 0;
    int word_count = 0;
    int byte_count = 0;
    size_t line_size = LINE_MAX;
    FILE *file_stream;
    char *line_buffer;

    printf("%d %d %d\n", line_count, word_count, byte_count);

    file_stream = fopen(filename, "r");

    if (file_stream == NULL) {
        printf("File count not be opened.\n");
    } else {
        line_buffer = (char *)malloc(line_size);
        while (getline(&line_buffer, &line_size, file_stream) != EOF) {
            if (lines_flag) {
                line_count++;
            }
            printf("LINE - %s\n", line_buffer);

            for (int i = -1; line_buffer[i] != '\n'; i++) {
                printf("%c ", line_buffer[i]);
                byte_count++;
                if (isspace(line_buffer[i]) && !isspace(line_buffer[i - 1])) {
                    word_count++;
                }
            }
            printf("\n");
            byte_count++; //Count last \n character
        }
        fclose(file_stream);
        
        printf("\n");
        printf("%d\n", line_count);
        printf("%d\n", word_count);
        printf("%d\n", byte_count);
        printf("\n");

        counts[0] = line_count;
        counts[1] = word_count;
        counts[2] = byte_count;
    }
    return counts;
}

调试输出

LINE - The Project Gutenberg eBook of The Art of War

    T h e   P r o j e c t   G u t e n b e r g   e B o o k   o f   T h e   A r t   o f   W a r 
LINE -     

         
LINE - This ebook is for the use of anyone anywhere in the United States and

 T h i s   e b o o k   i s   f o r   t h e   u s e   o f   a n y o n e   a n y w h e r e   i n   t h e   U n i t e d   S t a t e s   a n d 

3
24
130

目标文本文件内容

The Project Gutenberg eBook of The Art of War
    
This ebook is for the use of anyone anywhere in the United States and

问题根源分析

  1. 数组越界访问:for循环从i=-1开始,第一次读取line_buffer[-1]属于未定义行为——这会读取line_buffer内存地址之前的随机数据(就是你看到的前置空格),直接导致输出错位。
  2. 循环逻辑错误:循环条件line_buffer[i] != '\n'在i=-1时,读取的是越界内存而非当前行的内容,直到i递增到0后才开始读取当前行的有效字符,把越界的随机字符混入了当前行的输出。
  3. 单词统计逻辑漏洞:当i=0时,line_buffer[i-1]再次越界,且当前逻辑isspace(line_buffer[i]) && !isspace(line_buffer[i-1])会把空格当成单词计数的触发条件,完全不符合“单词由非空格字符分隔”的规则。
  4. 字节数重复计数:getline返回的字节数已经包含换行符,而代码中手动byte_count++又额外计数了一次换行符,导致字节数统计偏多。

修复方案

修正核心逻辑

#include <stdio.h>
#include <stdlib.h>
#include <stdbool.h>
#include <ctype.h>
#include <limits.h>

int *counter(const char *filename, bool bytes_flag, bool words_flag, bool lines_flag) {
    int *counts = malloc(sizeof(int) * 3);
    // 初始化计数为0
    counts[0] = counts[1] = counts[2] = 0;
    int line_count = 0;
    int word_count = 0;
    int byte_count = 0;
    size_t line_size = LINE_MAX;
    FILE *file_stream;
    char *line_buffer = NULL;
    ssize_t read_bytes;

    file_stream = fopen(filename, "r");
    if (file_stream == NULL) {
        perror("Failed to open file");
        return counts;
    }

    while ((read_bytes = getline(&line_buffer, &line_size, file_stream)) != EOF) {
        if (lines_flag) {
            line_count++;
        }
        // 直接用getline返回值统计字节数(包含换行符)
        if (bytes_flag) {
            byte_count += read_bytes;
        }

        printf("LINE - %s", line_buffer); // line_buffer已包含换行符,无需额外\n

        bool in_word = false;
        for (int i = 0; i < read_bytes; i++) {
            printf("%c ", line_buffer[i]);
            if (words_flag) {
                if (!isspace((unsigned char)line_buffer[i])) {
                    if (!in_word) {
                        word_count++;
                        in_word = true;
                    }
                } else {
                    in_word = false;
                }
            }
        }
        printf("\n");
    }

    fclose(file_stream);
    free(line_buffer); // 释放getline分配的内存,避免泄漏

    printf("\n统计结果:\n");
    printf("行数:%d\n", line_count);
    printf("单词数:%d\n", word_count);
    printf("字节数:%d\n", byte_count);

    counts[0] = line_count;
    counts[1] = word_count;
    counts[2] = byte_count;
    return counts;
}

关键修复点

  • 把for循环起始索引改为i=0,用getline返回的read_bytes作为循环边界,彻底避免越界访问。
  • 用in_word标志跟踪是否处于单词中,只有当从非单词状态切换到单词状态时,才增加单词计数,符合标准单词统计逻辑。
  • 直接使用getline的返回值统计字节数,避免手动计数错误。
  • 释放line_buffer的内存,修复内存泄漏问题。
  • 用perror打印文件打开错误,提供更清晰的错误信息。

内容的提问来源于stack exchange,提问作者Caleb Renfroe

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 06:14:54