You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言按行分割大文件程序行数不符问题修复求助

修复按指定行数准确分割文件的C代码

问题背景

现有一段用于按行分割大文件的C代码(chunk.c),指定按100行分割文件时,实际输出的文件行数出现偏差(如102、101行),无法准确按指定行数分割。

原代码

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
#include <fcntl.h>
#include <errno.h>

#define DEFAULT_PREFIX "x"
#define DEFAULT_CHUNK_SIZE 1000
#define ALPHABET_SIZE 26
#define MAX_DIGITS 2

void print_usage() {
    printf("Usage: chunk [-l line_count | -w word_count | -c character_count] [-p prefix] [-s suffix] [-f filename.txt | < filename.txt]\n");
}

int main(int argc, char *argv[]) {
    char *prefix = DEFAULT_PREFIX;
    int chunk_size = DEFAULT_CHUNK_SIZE;
    int suffix_start = 0;
    char *filename = NULL;

    // Parse command line arguments
    int opt;
    while ((opt = getopt(argc, argv, "l:p:s:f:")) != -1) {
        switch (opt) {
            case 'l':
                chunk_size = atoi(optarg);
                break;
  
            case 'p':
                prefix = optarg;
                break;
  
            case 's':
                suffix_start = atoi(optarg);
                break;
  
            case 'f':
                filename = optarg;
                break;
  
            default:
                print_usage();
                return 1;
        }
    }

    // Open input file
    int input_fd = STDIN_FILENO;
    if (filename != NULL) {
        input_fd = open(filename, O_RDONLY);
        if (input_fd == -1) {
            printf("Error: could not open file '%s': %s\n", filename, strerror(errno));
            return -1;
        }
    }

    // Read input file and write output files
    int line_count = 0;
    int chunk_count = 0;
    char suffix[MAX_DIGITS + 1];
    suffix[MAX_DIGITS] = '\0';
    int output_fd = -1;

    while (1) {
        if (line_count == 0) {
            // Close previous output file
            if (output_fd != -1) {
                close(output_fd);
                output_fd = -1;
            }

            // Open new output file (get new filename)
            snprintf(suffix, MAX_DIGITS + 1, "%02d", suffix_start + chunk_count);
            char *filename = malloc(strlen(prefix) + strlen(suffix) + 1);
            strcpy(filename, prefix);
            strcat(filename, suffix);
            
            output_fd = open(filename, O_WRONLY | O_CREAT | O_TRUNC, S_IRUSR | S_IWUSR | 
                        S_IRGRP | S_IWGRP | S_IROTH);
            if (output_fd == -1) {
                printf("Error: could not create file '%s': %s\n", filename, 
                        strerror(errno));
                return -1;
            }
            free(filename);

            chunk_count++;
        } // close if loop 

        // Read input
        char buffer[chunk_size];
        ssize_t bytes_read = read(input_fd, buffer, chunk_size);
        if (bytes_read == -1) {
            printf("Error: could not read input: %s\n", strerror(errno));
            return -1;
        }
        if (bytes_read == 0) {
            break;
        }
        
        // write output
        ssize_t bytes_written = write(output_fd, buffer, bytes_read);
        if (bytes_written == -1) {
            printf("Error: could not write output : %s\n", strerror(errno));
            return -1;
        }
    
        // Update line count
        for (int i = 0; i < bytes_written; i++) {
            if (buffer[i] == '\n') {
                line_count++;
            }
        }
        // Check if it's time to start a new chunk
        if (line_count >= chunk_size) {
            line_count = 0;
        }
    } // close while loop

    // Close input and output files
    if (input_fd != STDIN_FILENO) {
        close(input_fd);
    }
    if (output_fd != -1) {
        close(output_fd);
    }

    return 0;
} // close main

预期运行结果

$ chunk -l 100 -f z_answer.jok.txt -p part- -s 00
$ echo $?   # check exit status
0
$ wc *part* z_answer.jok.txt 
  100   669  4052 part-00
  100   725  4221 part-01
  100   551  3373 part-02
  100   640  3763 part-03
  100   588  3685 part-04
  100   544  3468 part-05
   90   473  3017 part-06
  690  4190 25579 z_answer.jok.txt
 1380  8380 51158 total

实际运行结果

$ chunk -l 100 -f z_answer.jok.txt -p part- -s 00
$ echo $?   # check exit status
0
$ wc *part* z_answer.jok.txt 
  102   675  4100 part-00
  101   745  4300 part-01
  100   554  3400 part-02
  101   640  3800 part-03
  103   609  3800 part-04
  100   534  3400 part-05
   83   434  2779 part-06
  690  4190 25579 z_answer.jok.txt
 1380  8381 51158 total

问题分析

原代码的核心缺陷:

  1. 缓冲区大小混淆:将chunk_size(行数)直接作为读取缓冲区的字节大小,导致每次读取的字节数完全不合理,可能一次读取多行或半行。
  2. 未处理跨缓冲区边界:当读取的缓冲区末尾是半行内容时,直接写入当前文件,后续读取的剩余内容会被计入下一个文件的行数,导致计数混乱。
  3. 行数截断逻辑缺失:当行数达到阈值时,直接重置计数,但没有截断当前缓冲区中超过阈值的部分,导致多写入了后续的行。

修复后的代码

#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
#include <fcntl.h>
#include <errno.h>

#define DEFAULT_PREFIX "x"
#define DEFAULT_CHUNK_SIZE 1000
#define MAX_DIGITS 2
#define BUFFER_SIZE 4096  // 使用固定字节缓冲区,符合系统页大小

void print_usage() {
    printf("Usage: chunk [-l line_count] [-p prefix] [-s suffix] [-f filename.txt | < filename.txt]\n");
}

int main(int argc, char *argv[]) {
    char *prefix = DEFAULT_PREFIX;
    int target_lines = DEFAULT_CHUNK_SIZE;
    int suffix_start = 0;
    char *filename = NULL;

    // Parse command line arguments
    int opt;
    while ((opt = getopt(argc, argv, "l:p:s:f:")) != -1) {
        switch (opt) {
            case 'l':
                target_lines = atoi(optarg);
                if (target_lines <= 0) {
                    printf("Error: line count must be positive\n");
                    return 1;
                }
                break;
  
            case 'p':
                prefix = optarg;
                break;
  
            case 's':
                suffix_start = atoi(optarg);
                break;
  
            case 'f':
                filename = optarg;
                break;
  
            default:
                print_usage();
                return 1;
        }
    }

    // Open input file
    int input_fd = STDIN_FILENO;
    if (filename != NULL) {
        input_fd = open(filename, O_RDONLY);
        if (input_fd == -1) {
            printf("Error: could not open file '%s': %s\n", filename, strerror(errno));
            return -1;
        }
    }

    // 保存跨缓冲区的剩余内容
    char remaining_buf[BUFFER_SIZE] = {0};
    size_t remaining_len = 0;
    int current_lines = 0;
    int chunk_count = 0;
    char suffix[MAX_DIGITS + 1];
    suffix[MAX_DIGITS] = '\0';
    int output_fd = -1;

    // 打开第一个输出文件
    if (output_fd == -1) {
        snprintf(suffix, MAX_DIGITS + 1, "%02d", suffix_start + chunk_count);
        char *out_filename = malloc(strlen(prefix) + strlen(suffix) + 1);
        strcpy(out_filename, prefix);
        strcat(out_filename, suffix);
        
        output_fd = open(out_filename, O_WRONLY | O_CREAT | O_TRUNC, S_IRUSR | S_IWUSR | 
                    S_IRGRP | S_IWGRP | S_IROTH);
        if (output_fd == -1) {
            printf("Error: could not create file '%s': %s\n", out_filename, strerror(errno));
            free(out_filename);
            return -1;
        }
        free(out_filename);
        chunk_count++;
    }

    while (1) {
        char buffer[BUFFER_SIZE];
        ssize_t bytes_read = read(input_fd, buffer, BUFFER_SIZE - remaining_len);
        if (bytes_read == -1) {
            printf("Error: could not read input: %s\n", strerror(errno));
            return -1;
        }
        if (bytes_read == 0) {
            // 处理剩余内容
            if (remaining_len > 0) {
                write(output_fd, remaining_buf, remaining_len);
            }
            break;
        }

        // 合并剩余内容和新读取的内容
        memcpy(remaining_buf + remaining_len, buffer, bytes_read);
        size_t total_len = remaining_len + bytes_read;
        size_t write_pos = 0;

        // 逐字符扫描换行符
        for (size_t i = 0; i < total_len; i++) {
            if (remaining_buf[i] == '\n') {
                current_lines++;
                // 达到目标行数,写入到当前换行符位置
                if (current_lines >= target_lines) {
                    write(output_fd, remaining_buf + write_pos, i - write_pos + 1);
                    write_pos = i + 1;
                    current_lines = 0;

                    // 关闭当前文件,打开新文件
                    close(output_fd);
                    snprintf(suffix, MAX_DIGITS + 1, "%02d", suffix_start + chunk_count);
                    char *out_filename = malloc(strlen(prefix) + strlen(suffix) + 1);
                    strcpy(out_filename, prefix);
                    strcat(out_filename, suffix);
                    
                    output_fd = open(out_filename, O_WRONLY | O_CREAT | O_TRUNC, S_IRUSR | S_IWUSR | 
                                S_IRGRP | S_IWGRP | S_IROTH);
                    if (output_fd == -1) {
                        printf("Error: could not create file '%s': %s\n", out_filename, strerror(errno));
                        free(out_filename);
                        return -1;
                    }
                    free(out_filename);
                    chunk_count++;
                }
            }
        }

        // 写入未达到换行阈值的内容
        if (write_pos < total_len) {
            write(output_fd, remaining_buf + write_pos, total_len - write_pos);
        }

        // 保存剩余的半行内容(如果有)
        remaining_len = 0;
        // 检查最后是否是半行
        if (remaining_buf[total_len - 1] != '\n') {
            // 从最后一个换行符之后复制到remaining_buf
            for (size_t i = total_len - 1; i >= 0; i--) {
                if (remaining_buf[i] == '\n') {
                    remaining_len = total_len - i - 1;
                    memcpy(remaining_buf, remaining_buf + i + 1, remaining_len);
                    break;
                }
                if (i == 0) {
                    remaining_len = total_len;
                    memcpy(remaining_buf, remaining_buf, remaining_len);
                    break;
                }
            }
        }
    }

    // Close input and output files
    if (input_fd != STDIN_FILENO) {
        close(input_fd);
    }
    if (output_fd != -1) {
        close(output_fd);
    }

    return 0;
}

关键修改说明

  1. 固定字节缓冲区:改用BUFFER_SIZE 4096作为读取缓冲区,避免行数和字节数混淆,提升IO效率。
  2. 剩余内容处理:新增remaining_buf和remaining_len保存跨缓冲区的半行内容,确保每行完整归属到对应的文件。
  3. 精确换行计数:逐字符扫描换行符,当达到指定行数时,截断写入并切换到新文件,确保每个文件严格不超过目标行数。
  4. 参数校验:新增对target_lines的正整数校验,避免无效输入。

内容的提问来源于stack exchange,提问作者springbook

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.27 06:12:02