You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C语言my_str_to_word_array函数内存错误与释放问题求助

C语言my_str_to_word_array函数内存错误排查

问题背景

开发C语言函数char **my_str_to_word_array(char *str),功能是将输入字符串按非打印ASCII字符分割,分割后的子串存入二维数组,分隔符不保留。

测试示例

char *test = "My name is John Doe.\nI have 0 GPA.\nI will survive." ;
char **array = my_str_to_word_array(test) ;

array[0] = "My name is John Doe." (零终止字符串)
array[1] = "I have 0 GPA." (零终止字符串)
array[2] = "I will survive." (零终止字符串)
array[3] = NULL

当前遇到的问题

  • 测试main()中,调用函数后执行printf,格式字符串被混入输出结果,推测存在内存越界读取;
  • 调用内存释放函数时触发double free or corruption (out)错误,程序崩溃。

实现代码

size_t get_words_number(char const *str)
{
    size_t count = 0;
    const char *i = str;
    while (*i != 0) {
        if (isprint(*i)) {
            count++;
        }
        while (*i != 0 && isprint(*i)) {
            i++;
        }
        i++;
    }
    return count;
}

char **free_corrupted_array(char **array, size_t i)
{
    size_t j = 0;
    while (j < i) {
        free(array[j]);
        j++;
    }
    free(array);
    return NULL;
}

char **fill_array(char **array, const char *str, size_t word_count)
{
    size_t word_size = 0, j = 0;
    const char *i = str;
    while (j < word_count) {
        while (*i != 0 && isprint(*i)) {
            word_size++;
            i++;
        }
        array[j] = strndup(i - word_size, word_size);
        if (!array[j]) {
            return free_corrupted_array(array, j);
        }
        word_size = 0;
        j++;
        while (!isprint(*i)) {
            i++;
        }
    }
    array[j] = NULL;
    return array;
}

char **my_str_to_word_array(char const *str)
{
    char **word_array = NULL;
    size_t word_count = 0;
    if (!str) {
        return NULL;
    }
    word_count = get_words_number(str);
    word_array = malloc(word_count * sizeof(char *));
    if (!word_array) {
        return NULL;
    }
    word_array = fill_array(word_array, str, word_count);
    return word_array;
}

内存释放函数

void my_free_word_array(char **word_array)
{
    if (!word_array) {
        return;
    }
    while (*word_array != NULL) {
        free(*word_array);
        word_array++;
    }
    free(word_array);
}

测试main函数

int main(int argc, char **argv)
{
    const char *test = "My name is John Doe.\nI have 0 GPA.\nI will survive.";
    char **word_array = my_str_to_word_array(test);
    while (*word_array != NULL) {
        printf("%s\n", *word_array);
        word_array++;
    }
    printf("Test print original size %lu\n", strlen(test));
    my_free_word_array(word_array);
    return 0;
}

程序输出

My name is John Doe.
I have 0 GPA.
I will survive.
Test print original size %lu
Test print original size 50
double free or corruption (out)
[1]    33429 IOT instruction (core dumped)  ./libmy

问题排查与修复

1. 内存越界导致printf格式串错乱

原因

my_str_to_word_array中分配数组时,仅申请了word_count * sizeof(char *)的空间,但函数要求数组末尾必须添加NULL哨兵,实际需要(word_count + 1) * sizeof(char *)的空间。当前分配空间不足,写入array[j] = NULL时会越界,破坏后续内存区域(包括printf格式串的存储)。

修复

修改my_str_to_word_array中的malloc语句:

word_array = malloc((word_count + 1) * sizeof(char *));

2. double free错误

原因

  • 指针偏移丢失原地址:main函数中遍历数组时执行word_array++,导致传入my_free_word_array的指针已不是原数组的起始地址,而是指向NULL的下一个位置。此时free(word_array)释放的不是malloc分配的原指针,触发内存错误。
  • 字符串边界处理不当:get_words_number和fill_array中,未检查指针是否到达字符串末尾就直接移动,导致越界访问*i(超出字符串的'\0')。

修复

  1. 保存原数组指针:在main函数中记录数组起始地址,避免遍历后丢失:
int main(int argc, char **argv)
{
    const char *test = "My name is John Doe.\nI have 0 GPA.\nI will survive.";
    char **word_array = my_str_to_word_array(test);
    char **original_array = word_array; // 保存原指针
    while (*word_array != NULL) {
        printf("%s\n", *word_array);
        word_array++;
    }
    printf("Test print original size %lu\n", strlen(test));
    my_free_word_array(original_array); // 传入原指针释放
    return 0;
}
  1. 修正get_words_number的边界逻辑:避免越界访问字符串末尾:
size_t get_words_number(char const *str)
{
    size_t count = 0;
    const char *i = str;
    while (*i != 0) {
        if (isprint((unsigned char)*i)) {
            count++;
            // 跳过当前单词的所有打印字符
            while (*i != 0 && isprint((unsigned char)*i)) {
                i++;
            }
        } else {
            // 跳过非打印字符
            i++;
        }
    }
    return count;
}
  1. 修正fill_array的边界逻辑:处理字符串末尾时避免越界:
char **fill_array(char **array, const char *str, size_t word_count)
{
    size_t word_size = 0, j = 0;
    const char *i = str;
    while (j < word_count) {
        // 前置跳过非打印字符
        while (*i != 0 && !isprint((unsigned char)*i)) {
            i++;
        }
        // 计算当前单词长度
        word_size = 0;
        while (*i != 0 && isprint((unsigned char)*i)) {
            word_size++;
            i++;
        }
        array[j] = strndup(i - word_size, word_size);
        if (!array[j]) {
            return free_corrupted_array(array, j);
        }
        j++;
    }
    array[j] = NULL;
    return array;
}

额外注意点

使用isprint时,需将传入的字符转为unsigned char类型,避免负数(如ASCII值>127的字符)导致未定义行为,示例中已添加(unsigned char)强制转换。

内容的提问来源于stack exchange,提问作者user20854333

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.06 04:35:56