如何从void*中提取各char*?C++未知长度日志拆分问询
拆分void*指向的多日志块为单个日志字符串
看你的日志格式,每条都是以YYYY/MM/DD HH:MM:SS这种日期时间开头的,这是关键分割标记——单条日志内部的换行不是分隔符,日志的起始特征才是分割依据。下面给两种可行的实现方式:
方法一:手动遍历匹配日志起始(高性能,无依赖)
这种方法不用正则,直接遍历缓冲区定位每个日志的起始位置,截取内容存入eachLog数组。修改你的测试代码如下:
#include <stdio.h> #include <cstdlib> #include <cstring> #include <stdlib.h> // 判断当前位置是否是日志起始(匹配YYYY/MM/DD格式开头) int is_log_start(const char* ptr) { // 简单校验:前4位数字+第5位/+第6-7位数字+第8位/+第9-10位数字 return (ptr[0] >= '0' && ptr[0] <= '9') && (ptr[1] >= '0' && ptr[1] <= '9') && (ptr[2] >= '0' && ptr[2] <= '9') && (ptr[3] >= '0' && ptr[3] <= '9') && (ptr[4] == '/') && (ptr[5] >= '0' && ptr[5] <= '9') && (ptr[6] >= '0' && ptr[6] <= '9') && (ptr[7] == '/') && (ptr[8] >= '0' && ptr[8] <= '9') && (ptr[9] >= '0' && ptr[9] <= '9'); } int main() { void *logs = malloc(1024); char eachLog[100][100]; int log_count = 0; if (logs == NULL) { printf("malloc failed\n"); return -1; } memcpy(logs, "2023/05/29 10:12:16 638377 [ debug] this is\n 1st log\n", 53); memcpy((char*)logs + 53, "2023/05/29 10:12:16 638378 [ err] this is 2st log\n", 52); memcpy((char*)logs + 105, "2023/05/29 10:12:16 638379 [ info] this is 3th log\n", 52); // 给缓冲区加结束符,防止越界访问 *((char*)logs + 53 + 52 + 52) = '\0'; char* buf_start = (char*)logs; char* current_pos = buf_start; char* prev_start = buf_start; // 第一条日志从开头开始 if (is_log_start(prev_start)) { log_count++; } // 遍历缓冲区,寻找后续日志起始 while (*current_pos != '\0') { current_pos++; // 确保剩余字符足够匹配日期格式,避免越界 if ((current_pos + 9) <= (buf_start + 1023) && is_log_start(current_pos)) { int len = current_pos - prev_start; // 复制到eachLog,限制长度防止越界 if (log_count < 100 && len < 100) { strncpy(eachLog[log_count - 1], prev_start, len); eachLog[log_count - 1][len] = '\0'; } else { printf("Log count or length exceeds eachLog limit\n"); } prev_start = current_pos; log_count++; } } // 处理最后一条日志 if (log_count > 0 && prev_start != current_pos) { int len = current_pos - prev_start; if (log_count - 1 < 100 && len < 100) { strncpy(eachLog[log_count - 1], prev_start, len); eachLog[log_count - 1][len] = '\0'; } } // 打印测试结果 for (int i = 0; i < log_count; i++) { printf("=== Log %d ===\n%s\n", i+1, eachLog[i]); } free(logs); logs = NULL; return 0; }
说明:
is_log_start函数可根据你的实际日志格式调整校验规则(比如年份位数、分隔符变化)。- 遍历过程中严格检查缓冲区边界,避免非法内存访问。
- 用
strncpy限制复制长度,确保不会超出eachLog单个元素的容量。
方法二:使用正则表达式(C++11及以上)
如果编译器支持C++11或更高版本,可以用<regex>库简化逻辑,代码更易读:
#include <stdio.h> #include <cstdlib> #include <cstring> #include <regex> #include <string> int main() { void *logs = malloc(1024); char eachLog[100][100]; int log_count = 0; if (logs == NULL) { printf("malloc failed\n"); return -1; } memcpy(logs, "2023/05/29 10:12:16 638377 [ debug] this is\n 1st log\n", 53); memcpy((char*)logs + 53, "2023/05/29 10:12:16 638378 [ err] this is 2st log\n", 52); memcpy((char*)logs + 105, "2023/05/29 10:12:16 638379 [ info] this is 3th log\n", 52); *((char*)logs + 53 + 52 + 52) = '\0'; std::string log_str((char*)logs); // 正则匹配日志起始的日期时间部分 std::regex log_start_re(R"(\d{4}/\d{2}/\d{2} \d{2}:\d{2}:\d{2})"); auto it = std::sregex_iterator(log_str.begin(), log_str.end(), log_start_re); auto end_it = std::sregex_iterator(); std::string prev_match; for (std::sregex_iterator i = it; i != end_it; ++i) { std::smatch match = *i; if (i == it) { prev_match = match.str(); continue; } // 截取上一个匹配到当前匹配之间的内容作为单条日志 std::string log_entry = log_str.substr(prev_match.data() - log_str.data(), match.position() - (prev_match.data() - log_str.data())); if (log_count < 100 && log_entry.size() < 100) { strcpy(eachLog[log_count], log_entry.c_str()); log_count++; } prev_match = match.str(); } // 处理最后一条日志 if (it != end_it) { std::smatch last_match = *(--end_it); std::string log_entry = log_str.substr(last_match.position()); if (log_count < 100 && log_entry.size() < 100) { strcpy(eachLog[log_count], log_entry.c_str()); log_count++; } } // 打印测试结果 for (int i = 0; i < log_count; i++) { printf("=== Log %d ===\n%s\n", i+1, eachLog[i]); } free(logs); logs = NULL; return 0; }
说明:
- 正则表达式可根据日志实际格式修改(比如加入进程ID、日志级别的匹配规则)。
- 用
std::sregex_iterator遍历所有日志起始位置,自动分割内容。 - 同样需要注意
eachLog的容量限制,避免数组越界。
通用注意事项:
- 必须确保
void*指向的缓冲区以\0结尾,或者你已知缓冲区总长度,否则遍历会出现非法内存访问。 - 如果单条日志长度可能超过99,需要调整
eachLog的元素大小,或者改用动态分配的字符串(比如std::string)。
内容的提问来源于stack exchange,提问作者Yongqi Z
相关产品推荐
相关产品推荐

