You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言逐字节读文件并处理文本:去标点转小写存入字符数组

C语言逐字节处理文件并提取单词实现方案

需求说明

  • 逐字节读取文件内容,每次读取一个字节后执行后续操作
  • 输出读取到的内容片段(例如示例中的To Be, or)
  • 去除内容中的空格与标点符号,将所有字符转为小写,拆分出单个单词
  • 将处理后的单词存入char*类型数组(例如char *words[] = { "to", "be", "or" })

示例说明:若文件内容为To be, or not to be, that is the question:,需先读取To Be, or并输出,处理后得到to、be、or存入数组,再继续读取not to be,并重复流程。

现有代码问题分析

当前代码存在以下关键问题,无法满足需求:

  1. 一次性读取整个文件到缓冲区,未实现逐字节循环处理的逻辑
  2. 初始化char* words[a]时a=0,数组长度为0,无法存储单词
  3. 缺少字符转小写、去除标点、拆分单词的核心处理逻辑
  4. 未按要求分片段输出内容并处理

完整解决方案代码

#include <stdio.h>
#include <ctype.h>
#include <stdlib.h>
#include <string.h>

#define SEGMENT_SIZE 20  // 定义每个输出片段的最大长度
#define INIT_WORDS_CAPACITY 10  // 单词数组初始容量

int main() {
    FILE* fp = fopen("finalread.c", "r");
    if (!fp) {
        perror("Failed to open file");
        return 1;
    }

    char segment[SEGMENT_SIZE + 1] = {0};  // 存储当前输出片段
    int seg_idx = 0;
    char current_word[50] = {0};  // 存储正在构建的单词
    int word_idx = 0;
    char** words = malloc(INIT_WORDS_CAPACITY * sizeof(char*));  // 动态单词数组
    int word_count = 0;
    int words_capacity = INIT_WORDS_CAPACITY;

    int c;
    while ((c = fgetc(fp)) != EOF) {
        // 将当前字符加入片段缓冲区
        if (seg_idx < SEGMENT_SIZE) {
            segment[seg_idx++] = c;
            segment[seg_idx] = '\0';
        } else {
            // 片段已满,输出并重置
            printf("读取到的片段: %s\n", segment);
            seg_idx = 0;
            memset(segment, 0, sizeof(segment));
            segment[seg_idx++] = c;
            segment[seg_idx] = '\0';
        }

        // 处理字符,构建单词
        if (isalpha(c)) {
            // 字母转小写,加入当前单词
            current_word[word_idx++] = tolower(c);
            current_word[word_idx] = '\0';
        } else {
            // 遇到非字母字符,若当前有未存储的单词则处理
            if (word_idx > 0) {
                // 扩容单词数组
                if (word_count >= words_capacity) {
                    words_capacity *= 2;
                    words = realloc(words, words_capacity * sizeof(char*));
                    if (!words) {
                        perror("Failed to reallocate memory");
                        return 1;
                    }
                }
                // 分配内存存储单词并复制
                words[word_count] = malloc(strlen(current_word) + 1);
                strcpy(words[word_count], current_word);
                word_count++;
                // 重置当前单词缓冲区
                word_idx = 0;
                memset(current_word, 0, sizeof(current_word));
            }

            // 如果片段已满或者遇到换行符,输出当前片段
            if (seg_idx == SEGMENT_SIZE || c == '\n') {
                printf("读取到的片段: %s\n", segment);
                seg_idx = 0;
                memset(segment, 0, sizeof(segment));
            }
        }
    }

    // 处理文件末尾可能剩余的单词
    if (word_idx > 0) {
        if (word_count >= words_capacity) {
            words_capacity *= 2;
            words = realloc(words, words_capacity * sizeof(char*));
        }
        words[word_count] = malloc(strlen(current_word) + 1);
        strcpy(words[word_count], current_word);
        word_count++;
    }

    // 输出剩余的片段
    if (seg_idx > 0) {
        printf("读取到的片段: %s\n", segment);
    }

    // 输出处理后的单词数组
    printf("\n处理后的单词数组:\n");
    for (int i = 0; i < word_count; i++) {
        printf("words[%d] = \"%s\"\n", i, words[i]);
        free(words[i]);  // 释放单个单词的内存
    }
    free(words);  // 释放单词数组的内存
    fclose(fp);

    return 0;
}

代码核心逻辑说明

  1. 逐字节读取:使用fgetc逐字节读取文件,直到文件末尾
  2. 片段输出:用segment缓冲区积累字符,达到SEGMENT_SIZE或遇到换行符时输出当前片段
  3. 单词处理:
    • 仅保留字母字符,转为小写后存入current_word缓冲区
    • 遇到非字母字符时,将current_word中的内容存入动态扩容的words数组
  4. 内存管理:单词数组初始分配固定容量,不足时通过realloc扩容;程序结束前释放所有分配的内存,避免泄漏

输出示例(对应示例文件内容)

读取到的片段: To be, or not to 
读取到的片段: be, that is the q
读取到的片段: uestion:

处理后的单词数组:
words[0] = "to"
words[1] = "be"
words[2] = "or"
words[3] = "not"
words[4] = "to"
words[5] = "be"
words[6] = "that"
words[7] = "is"
words[8] = "the"
words[9] = "question"

内容的提问来源于stack exchange,提问作者Jaya Vardhini Akurathi

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.09 09:40:32