C语言实现字符串编码函数时结果串写入内存溢出问题排查
编码函数实现问题说明
需要实现的编码函数接收原始字符串、结果字符串二级指针两个入参,按规则完成编码后将结果写入指定内存空间,编码规则如下:
- 数字字符重复
n+1次写入结果串,例如字符'2'对应输出为"222" - 大写字母先转换为对应小写字母,再将该小写字母的ASCII值按数字位逆序写入结果串,例如
'A'转换为小写'a'的ASCII值为97,逆序后写入内容为"79" - 小写字母先转换为字母顺序逆置的对应大写字母(即
'a'对应'Z'、'b'对应'Y',以此类推),再将该大写字母的ASCII值按数字位逆序写入结果串,例如'a'对应大写'Z'的ASCII值为90,逆序后写入内容为"09"
当前已梳理完编码逻辑,但写入结果串时触发内存溢出,不确定是内存分配逻辑还是写入逻辑存在问题,附现有实现代码与宏定义如下:
现有宏定义
#define BOT_DIGIT 48 #define TOP_DIGIT 57 #define BOT_UPPER 65 #define TOP_UPPER 90 #define BOT_LOWER 97 #define TOP_LOWER 122 #define C 99
现有函数实现
int get_codedstring(char *orig_str, char **coded_str) { // figuring out the minimal length needed for result_str int len = 1; // minimal length for string that has nothing in it exept'\0' int i = 0; while (orig_str[i] != '\0') { if ((int)orig_str[i] >= BOT_DIGIT && (int)orig_str[i] <= TOP_DIGIT) // digit len += (int)orig_str[i] - BOT_DIGIT + 1; else if ((int)orig_str[i] >= BOT_UPPER && (int)orig_str[i] <= TOP_UPPER) // uppercase { if ((int)orig_str[i] >= BOT_LOWER && (int)orig_str[i] <= C) len += 2; else len += 3; } else // lowercase len += 2; i++; } *coded_str = malloc(sizeof(char) * len); // allocating mem for res_str // encoding i = 0; int j = 0; int temp; while (orig_str[i] != '\0') { if ((int)orig_str[i] >= BOT_DIGIT && (int)orig_str[i] <= TOP_DIGIT) // digit { temp = (int)orig_str[i] - BOT_DIGIT + 1; for (j; j < temp; j++) { *(coded_str[i + j]) = orig_str[i]; } j = i + j; } else if ((int)orig_str[i] >= BOT_UPPER && (int)orig_str[i] <= TOP_UPPER) // uppercase { temp = (int)orig_str[i] + 32; // switch to lowercase // add to coded_str in reverse digit order if (((int)orig_str[i] + 32) >= BOT_LOWER && ((int)orig_str[i] + 32) <= C) { // a=97,b=98,c=99 the rest need 3 spaces *coded_str[i + j] = (char)(temp % 10) + BOT_DIGIT; j++; *coded_str[i + j] = (char)(temp / 10) + BOT_DIGIT; j++; } else { *coded_str[i + j] = (char)(temp % 100) + BOT_DIGIT; j++; temp /= 10; *coded_str[i + j] = (char)(temp % 10) + BOT_DIGIT; j++; *coded_str[i + j] = (char)(temp / 10) + BOT_DIGIT; j++; } } else // lowercase { temp = (int)orig_str[i] - 32; // switch to lower case temp = TOP_UPPER - (temp - BOT_UPPER); // reverse order *coded_str[i + j] = (char)(temp % 10) + BOT_DIGIT; j++; *coded_str[i + j] = (char)(temp / 10) + BOT_DIGIT; j++; } i++; } return len; }
问题根因
内存溢出由多处逻辑错误共同导致:
- 二级指针解引用错误:
coded_str是二级指针,指向存储结果串地址的指针变量,*coded_str才是malloc返回的结果数组首地址。现有代码直接写coded_str[i+j]、*coded_str[i+j],因为下标运算符优先级高于解引用,实际等价于*(coded_str[i+j]),相当于把coded_str当成字符指针数组索引,访问的完全不是申请的内存空间,直接触发非法访问。 - 结果串索引计算错误:
i是原始字符串的遍历下标,j是结果串的已写入长度,结果串的写入位置只需要用j偏移即可。现有代码用i+j当偏移量,每处理一个原串字符就多偏移i个字节,写入位置快速超出申请的内存范围,必然溢出。 - 内存长度计算错误:判断大写字母对应写入长度时,拿原始大写字符的ASCII值和
BOT_LOWER(97)比较,而大写字母ASCII范围是65-90,该判断条件永远不成立,导致A、B、C三个大写字母本来只需要2位存储,都按3位计算,和索引偏移错误叠加后进一步加剧越界。 - 写入逻辑错误:处理数字时for循环初始值未重置,循环结束后
j = i + j的偏移计算完全混乱;处理三位ASCII值的大写字母时,用temp%100取个位,得到的是十位加个位的数值,加BOT_DIGIT后根本不是合法数字字符;所有分支处理完后没有给结果串添加末尾的'\0'终止符,字符串不合法,后续读取也会越界。
修正后实现
#include <stdlib.h> #define BOT_DIGIT 48 #define TOP_DIGIT 57 #define BOT_UPPER 65 #define TOP_UPPER 90 #define BOT_LOWER 97 #define TOP_LOWER 122 int get_codedstring(char *orig_str, char **coded_str) { int len = 1; // 预留'\0'位置 int i = 0; // 精确计算所需内存长度 while (orig_str[i] != '\0') { if (orig_str[i] >= BOT_DIGIT && orig_str[i] <= TOP_DIGIT) { len += orig_str[i] - BOT_DIGIT + 1; } else if (orig_str[i] >= BOT_UPPER && orig_str[i] <= TOP_UPPER) { int lower_ascii = orig_str[i] + 32; // a(97)~c(99)为两位,d(100)~z(122)为三位 len += (lower_ascii <= 99) ? 2 : 3; } else if (orig_str[i] >= BOT_LOWER && orig_str[i] <= TOP_LOWER) { len += 2; } i++; } *coded_str = (char*)malloc(sizeof(char) * len); if (*coded_str == NULL) return -1; // 增加内存分配失败判断 i = 0; int j = 0; // j始终指向结果串下一个可写位置 int temp; while (orig_str[i] != '\0') { if (orig_str[i] >= BOT_DIGIT && orig_str[i] <= TOP_DIGIT) { // 处理数字 int repeat = orig_str[i] - BOT_DIGIT + 1; for (int k = 0; k < repeat; k++) { (*coded_str)[j++] = orig_str[i]; } } else if (orig_str[i] >= BOT_UPPER && orig_str[i] <= TOP_UPPER) { // 处理大写字母 temp = orig_str[i] + 32; if (temp <= 99) { // 两位ASCII逆序写入 (*coded_str)[j++] = (temp % 10) + BOT_DIGIT; (*coded_str)[j++] = (temp / 10) + BOT_DIGIT; } else { // 三位ASCII逆序写入 (*coded_str)[j++] = (temp % 10) + BOT_DIGIT; temp /= 10; (*coded_str)[j++] = (temp % 10) + BOT_DIGIT; temp /= 10; (*coded_str)[j++] = (temp % 10) + BOT_DIGIT; } } else if (orig_str[i] >= BOT_LOWER && orig_str[i] <= TOP_LOWER) { // 处理小写字母 temp = orig_str[i] - 32; temp = TOP_UPPER - (temp - BOT_UPPER); (*coded_str)[j++] = (temp % 10) + BOT_DIGIT; (*coded_str)[j++] = (temp / 10) + BOT_DIGIT; } i++; } (*coded_str)[j] = '\0'; // 补字符串终止符 return len; }
内容的提问来源于stack exchange,提问作者IgorK
相关产品推荐
相关产品推荐

