C程序无法识别echo命令创建文件中的子串问题求助
子串统计C程序在echo创建文件时失效的排查与解决
我开发了一款统计指定文件中子串出现次数的C程序,手动创建并填充内容的文件能正常统计,但用echo命令创建的文件,明明文本编辑器里能看到目标子串,程序却检测不到。以下是简化版代码、操作命令及输出:
简化版程序代码
#include <stdio.h> #include <stdlib.h> #include <string.h> #include <unistd.h> #include <sys/types.h> #define BUFFER_SIZE 1024 int num_substrings = 0; int use_systemcall = 0; void search_file(char *filename, char *substring) { // Open the file with the given filename in read mode FILE *file = fopen(filename, "r"); // Check if the file was successfully opened if (file == NULL) { // Print an error message and exit the program with an error code fprintf(stderr, "Error: could not open file '%s'\n", filename); exit(1); } int count = 0; char buffer[BUFFER_SIZE]; char *line; size_t len = 0; ssize_t read; // Read the file line by line until the end while ((read = getline(&line, &len, file)) != -1) { // Skip the last line if it is empty if (read == 1 && line[0] == '\n') { continue; } // Strip any newline characters from the end of the line if (line[read - 1] == '\n') { line[read - 1] = '\0'; read--; } // Find the first occurrence of the given substring in the current line char *match = strstr(line, substring); // While there are still occurrences of the substring in the current line while (match != NULL) { // Increment the counter and find the next occurrence of the substring count++; match = strstr(match + 1, substring); } } // Close the file fclose(file); // Print the number of occurrences of the substring found in the file printf("Found %d occurrences of substring '%s' in file '%s'\n", count, substring, filename); } int main(int argc, char *argv[]) { // Get the filename from the first command-line argument char *filename = argv[1]; // Initialize an array to store the substrings and a counter for the number of substrings char substrings[10][100]; int num_substrings = 0; // Loop through the remaining command-line arguments (starting from the second one) for (int i = 2; i < argc; i++) { // Copy the current argument (substring) into the substrings array strcpy(substrings[num_substrings], argv[i]); // Increment the counter for the number of substrings num_substrings++; } // Ask the user if they want to use a system call printf("Do you want to use system call? (y/n): "); char answer[10]; fgets(answer, 10, stdin); // Check if the user answered yes (y or Y) and set the use_systemcall variable accordingly int use_systemcall = 0; if (answer[0] == 'y' || answer[0] == 'Y') { use_systemcall = 1; } printf("Filename: %s\n", filename); printf("Substrings: "); for (int i = 0; i < num_substrings; i++) { printf("%s ", substrings[i]); } printf("\n"); // Open the file for reading FILE *file = fopen(filename, "rb"); if (file == NULL) { printf("Error: Cannot open file %s\n", filename); return 1; } // Initialize a buffer to read the file in blocks of 100 characters char buffer[101]; // Loop through each substring and search for it in the file for (int i = 0; i < num_substrings; i++) { // Reset the file pointer to the beginning of the file fseek(file, 0, SEEK_SET); // Initialize a counter for the number of occurrences of the substring int count = 0; // Loop through the file in blocks of 100 characters while (fread(buffer, sizeof(char), 100, file) > 0) { // Add a null terminator at the end of the buffer buffer[100] = '\0'; // Search for the substring in the buffer char *result = strstr(buffer, substrings[i]); // If the substring is found, increment the count while (result != NULL) { count++; // Move the result pointer to the next character after the match result++; // Search for the substring again starting from the result pointer result = strstr(result, substrings[i]); } } // Print the number of occurrences of the substring printf("'%s' appears %d times in the file.\n", substrings[i], count); } return 0; }
操作命令
echo "hello world" > foo.txt # 创建文件 ./substring_search foo.txt world # 搜索子串
程序输出
'world' appears 0 times in the file.
问题根源
核心问题是:你已经写好了正确的search_file函数,但在main中完全没有调用它,反而实现了一套有缺陷的独立读取统计逻辑:
search_file函数会逐行读取文件,自动去除每行末尾的换行符,能正确匹配echo生成文件中的world;但main里的代码完全忽略了这个函数,自己用fread按块读取文件。- 另外,
main中的块读取逻辑还存在跨块匹配丢失的问题(如果子串刚好在两个100字节块的边界,会被漏统计),不过在这个案例中,因为文件太小,这个问题没暴露,但核心是你没启用正确的逻辑。
解决方法
最优方案:调用已实现的search_file函数
直接修改main函数的后半部分,替换掉那套有问题的块读取逻辑,调用你已经写好的search_file函数:
printf("Filename: %s\n", filename); printf("Substrings: "); for (int i = 0; i < num_substrings; i++) { printf("%s ", substrings[i]); } printf("\n"); // 替换原有的文件读取统计代码,调用正确的search_file函数 for (int i = 0; i < num_substrings; i++) { search_file(filename, substrings[i]); } // 删除原有的fopen、fread等相关代码 return 0;
修改后重新编译运行,echo生成的文件就能正确统计出world的出现次数了。
可选方案:修复main中的块读取逻辑
如果你坚持要用块读取的方式,需要修复跨块匹配的问题,同时优化搜索逻辑:
// 修复后的单块读取搜索逻辑示例(针对单个子串) int sub_len = strlen(substrings[i]); char overlap_buf[100] = {0}; // 保存上一次读取末尾可能的重叠部分 int overlap_len = 0; int count = 0; fseek(file, 0, SEEK_SET); while (1) { char buffer[100]; int read_bytes = fread(buffer, sizeof(char), 100, file); if (read_bytes == 0) break; // 拼接重叠部分和当前读取的内容 char combined[200]; memcpy(combined, overlap_buf, overlap_len); memcpy(combined + overlap_len, buffer, read_bytes); combined[overlap_len + read_bytes] = '\0'; // 搜索所有匹配 char *result = strstr(combined, substrings[i]); while (result != NULL) { count++; result = strstr(result + 1, substrings[i]); } // 更新重叠部分:保留最后sub_len-1个字符,防止跨块匹配丢失 if (overlap_len + read_bytes >= sub_len - 1) { overlap_len = sub_len - 1; memcpy(overlap_buf, combined + (overlap_len + read_bytes - overlap_len), overlap_len); } else { overlap_len += read_bytes; memcpy(overlap_buf + (overlap_len - read_bytes), buffer, read_bytes); } } printf("'%s' appears %d times in the file.\n", substrings[i], count);
内容的提问来源于stack exchange,提问作者Moeez
相关产品推荐
相关产品推荐

