C语言按行分割大文件程序行数不符问题修复求助
修复按指定行数准确分割文件的C代码
问题背景
现有一段用于按行分割大文件的C代码(chunk.c),指定按100行分割文件时,实际输出的文件行数出现偏差(如102、101行),无法准确按指定行数分割。
原代码
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <unistd.h> #include <fcntl.h> #include <errno.h> #define DEFAULT_PREFIX "x" #define DEFAULT_CHUNK_SIZE 1000 #define ALPHABET_SIZE 26 #define MAX_DIGITS 2 void print_usage() { printf("Usage: chunk [-l line_count | -w word_count | -c character_count] [-p prefix] [-s suffix] [-f filename.txt | < filename.txt]\n"); } int main(int argc, char *argv[]) { char *prefix = DEFAULT_PREFIX; int chunk_size = DEFAULT_CHUNK_SIZE; int suffix_start = 0; char *filename = NULL; // Parse command line arguments int opt; while ((opt = getopt(argc, argv, "l:p:s:f:")) != -1) { switch (opt) { case 'l': chunk_size = atoi(optarg); break; case 'p': prefix = optarg; break; case 's': suffix_start = atoi(optarg); break; case 'f': filename = optarg; break; default: print_usage(); return 1; } } // Open input file int input_fd = STDIN_FILENO; if (filename != NULL) { input_fd = open(filename, O_RDONLY); if (input_fd == -1) { printf("Error: could not open file '%s': %s\n", filename, strerror(errno)); return -1; } } // Read input file and write output files int line_count = 0; int chunk_count = 0; char suffix[MAX_DIGITS + 1]; suffix[MAX_DIGITS] = '\0'; int output_fd = -1; while (1) { if (line_count == 0) { // Close previous output file if (output_fd != -1) { close(output_fd); output_fd = -1; } // Open new output file (get new filename) snprintf(suffix, MAX_DIGITS + 1, "%02d", suffix_start + chunk_count); char *filename = malloc(strlen(prefix) + strlen(suffix) + 1); strcpy(filename, prefix); strcat(filename, suffix); output_fd = open(filename, O_WRONLY | O_CREAT | O_TRUNC, S_IRUSR | S_IWUSR | S_IRGRP | S_IWGRP | S_IROTH); if (output_fd == -1) { printf("Error: could not create file '%s': %s\n", filename, strerror(errno)); return -1; } free(filename); chunk_count++; } // close if loop // Read input char buffer[chunk_size]; ssize_t bytes_read = read(input_fd, buffer, chunk_size); if (bytes_read == -1) { printf("Error: could not read input: %s\n", strerror(errno)); return -1; } if (bytes_read == 0) { break; } // write output ssize_t bytes_written = write(output_fd, buffer, bytes_read); if (bytes_written == -1) { printf("Error: could not write output : %s\n", strerror(errno)); return -1; } // Update line count for (int i = 0; i < bytes_written; i++) { if (buffer[i] == '\n') { line_count++; } } // Check if it's time to start a new chunk if (line_count >= chunk_size) { line_count = 0; } } // close while loop // Close input and output files if (input_fd != STDIN_FILENO) { close(input_fd); } if (output_fd != -1) { close(output_fd); } return 0; } // close main
预期运行结果
$ chunk -l 100 -f z_answer.jok.txt -p part- -s 00 $ echo $? # check exit status 0 $ wc *part* z_answer.jok.txt 100 669 4052 part-00 100 725 4221 part-01 100 551 3373 part-02 100 640 3763 part-03 100 588 3685 part-04 100 544 3468 part-05 90 473 3017 part-06 690 4190 25579 z_answer.jok.txt 1380 8380 51158 total
实际运行结果
$ chunk -l 100 -f z_answer.jok.txt -p part- -s 00 $ echo $? # check exit status 0 $ wc *part* z_answer.jok.txt 102 675 4100 part-00 101 745 4300 part-01 100 554 3400 part-02 101 640 3800 part-03 103 609 3800 part-04 100 534 3400 part-05 83 434 2779 part-06 690 4190 25579 z_answer.jok.txt 1380 8381 51158 total
问题分析
原代码的核心缺陷:
- 缓冲区大小混淆:将
chunk_size(行数)直接作为读取缓冲区的字节大小,导致每次读取的字节数完全不合理,可能一次读取多行或半行。 - 未处理跨缓冲区边界:当读取的缓冲区末尾是半行内容时,直接写入当前文件,后续读取的剩余内容会被计入下一个文件的行数,导致计数混乱。
- 行数截断逻辑缺失:当行数达到阈值时,直接重置计数,但没有截断当前缓冲区中超过阈值的部分,导致多写入了后续的行。
修复后的代码
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <unistd.h> #include <fcntl.h> #include <errno.h> #define DEFAULT_PREFIX "x" #define DEFAULT_CHUNK_SIZE 1000 #define MAX_DIGITS 2 #define BUFFER_SIZE 4096 // 使用固定字节缓冲区,符合系统页大小 void print_usage() { printf("Usage: chunk [-l line_count] [-p prefix] [-s suffix] [-f filename.txt | < filename.txt]\n"); } int main(int argc, char *argv[]) { char *prefix = DEFAULT_PREFIX; int target_lines = DEFAULT_CHUNK_SIZE; int suffix_start = 0; char *filename = NULL; // Parse command line arguments int opt; while ((opt = getopt(argc, argv, "l:p:s:f:")) != -1) { switch (opt) { case 'l': target_lines = atoi(optarg); if (target_lines <= 0) { printf("Error: line count must be positive\n"); return 1; } break; case 'p': prefix = optarg; break; case 's': suffix_start = atoi(optarg); break; case 'f': filename = optarg; break; default: print_usage(); return 1; } } // Open input file int input_fd = STDIN_FILENO; if (filename != NULL) { input_fd = open(filename, O_RDONLY); if (input_fd == -1) { printf("Error: could not open file '%s': %s\n", filename, strerror(errno)); return -1; } } // 保存跨缓冲区的剩余内容 char remaining_buf[BUFFER_SIZE] = {0}; size_t remaining_len = 0; int current_lines = 0; int chunk_count = 0; char suffix[MAX_DIGITS + 1]; suffix[MAX_DIGITS] = '\0'; int output_fd = -1; // 打开第一个输出文件 if (output_fd == -1) { snprintf(suffix, MAX_DIGITS + 1, "%02d", suffix_start + chunk_count); char *out_filename = malloc(strlen(prefix) + strlen(suffix) + 1); strcpy(out_filename, prefix); strcat(out_filename, suffix); output_fd = open(out_filename, O_WRONLY | O_CREAT | O_TRUNC, S_IRUSR | S_IWUSR | S_IRGRP | S_IWGRP | S_IROTH); if (output_fd == -1) { printf("Error: could not create file '%s': %s\n", out_filename, strerror(errno)); free(out_filename); return -1; } free(out_filename); chunk_count++; } while (1) { char buffer[BUFFER_SIZE]; ssize_t bytes_read = read(input_fd, buffer, BUFFER_SIZE - remaining_len); if (bytes_read == -1) { printf("Error: could not read input: %s\n", strerror(errno)); return -1; } if (bytes_read == 0) { // 处理剩余内容 if (remaining_len > 0) { write(output_fd, remaining_buf, remaining_len); } break; } // 合并剩余内容和新读取的内容 memcpy(remaining_buf + remaining_len, buffer, bytes_read); size_t total_len = remaining_len + bytes_read; size_t write_pos = 0; // 逐字符扫描换行符 for (size_t i = 0; i < total_len; i++) { if (remaining_buf[i] == '\n') { current_lines++; // 达到目标行数,写入到当前换行符位置 if (current_lines >= target_lines) { write(output_fd, remaining_buf + write_pos, i - write_pos + 1); write_pos = i + 1; current_lines = 0; // 关闭当前文件,打开新文件 close(output_fd); snprintf(suffix, MAX_DIGITS + 1, "%02d", suffix_start + chunk_count); char *out_filename = malloc(strlen(prefix) + strlen(suffix) + 1); strcpy(out_filename, prefix); strcat(out_filename, suffix); output_fd = open(out_filename, O_WRONLY | O_CREAT | O_TRUNC, S_IRUSR | S_IWUSR | S_IRGRP | S_IWGRP | S_IROTH); if (output_fd == -1) { printf("Error: could not create file '%s': %s\n", out_filename, strerror(errno)); free(out_filename); return -1; } free(out_filename); chunk_count++; } } } // 写入未达到换行阈值的内容 if (write_pos < total_len) { write(output_fd, remaining_buf + write_pos, total_len - write_pos); } // 保存剩余的半行内容(如果有) remaining_len = 0; // 检查最后是否是半行 if (remaining_buf[total_len - 1] != '\n') { // 从最后一个换行符之后复制到remaining_buf for (size_t i = total_len - 1; i >= 0; i--) { if (remaining_buf[i] == '\n') { remaining_len = total_len - i - 1; memcpy(remaining_buf, remaining_buf + i + 1, remaining_len); break; } if (i == 0) { remaining_len = total_len; memcpy(remaining_buf, remaining_buf, remaining_len); break; } } } } // Close input and output files if (input_fd != STDIN_FILENO) { close(input_fd); } if (output_fd != -1) { close(output_fd); } return 0; }
关键修改说明
- 固定字节缓冲区:改用
BUFFER_SIZE 4096作为读取缓冲区,避免行数和字节数混淆,提升IO效率。 - 剩余内容处理:新增
remaining_buf和remaining_len保存跨缓冲区的半行内容,确保每行完整归属到对应的文件。 - 精确换行计数:逐字符扫描换行符,当达到指定行数时,截断写入并切换到新文件,确保每个文件严格不超过目标行数。
- 参数校验:新增对
target_lines的正整数校验,避免无效输入。
内容的提问来源于stack exchange,提问作者springbook
相关产品推荐
相关产品推荐

