You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

IA-32 x86汇编实现矩阵乘积时栈缓冲区溢出问题排查

WORD矩阵乘积程序栈缓冲区溢出问题解决建议

我需要实现两个WORD(short int)矩阵的DWORD(int)乘积运算,当前程序在所有测试场景下输出结果正确,但打印完成后会因栈基缓冲区溢出崩溃,相关代码及修复方案如下:

原问题代码

#include <stdio.h>

void main()
{
    // Number of rows of the first matrix
    unsigned int m = 3;
    // Number of columns of the first matrix
    unsigned int n = 2;
    // Number of columns of the second matrix
    unsigned int k = 4;
    // First matrix
    short int mat1[] = { -1,-2, 4,5, 4,-2 };
    // Second matrix
    short int mat2[] = { 2,0,0,0, 0,2,0,0 };
    // Result matrix
    int mat3[1024];

    __asm {
        // Initialization of the row index to 0
        mov eax, 0          // i = 0

        // Outer loop for the rows of the result matrix
        esterno_ciclo:
        // Check if we have reached the last row of the matrix
        cmp eax, m
            jge fine_ciclo_esterno  // If i >= m, exit the outer loop

            // Initialization of the column index to 0
            mov ebx, 0          // j = 0

            // Inner loop for the columns of the result matrix
            interno_ciclo:
        // Check if we have reached the last column of the matrix
        cmp ebx, k
            jge fine_ciclo_interno  // If j >= k, exit the inner loop

            // Initialization of the sum to 0 for each new cell of the result matrix
            mov dword ptr[ebp - 4], 0  // sum = 0

            // Initialization of the column index of the first matrix to 0
            mov ecx, 0  // l = 0

            // Loop to multiply the elements of the matrices and sum the results
            ciclo_medio:
        // Check if we have reached the last column of the first matrix
        cmp ecx, n
            jge fine_ciclo_medio  // If l >= n, exit the middle loop

            // Calculate the addresses of the current elements of the matrices
            mov edx, eax                      // edx = i
            imul edx, n                       // edx = i * n
            add edx, ecx                      // edx = i * n + l
            movsx esi, word ptr[mat1 + edx * 2]  // esi = mat1[i * n + l]

            mov edx, ecx                      // edx = l
            imul edx, k                       // edx = l * k
            add edx, ebx                      // edx = l * k + j
            movsx edi, word ptr[mat2 + edx * 2]  // edi = mat2[l * k + j]

            // Multiply the elements of the matrices and add to the sum
            imul esi, edi                     // esi = mat1[i * n + l] * mat2[l * k + j]
            add dword ptr[ebp - 4], esi     // sum += mat1[i * n + l] * mat2[l * k + j]

            // Increment l
            inc ecx
            jmp ciclo_medio

            fine_ciclo_medio :

        // Save the sum in the result matrix
        mov edx, eax                   // edx = i
            imul edx, k                    // edx = i * k
            add edx, ebx                   // edx = i * k + j
            mov esi, dword ptr[ebp - 4]  // esi = sum
            mov dword ptr[mat3 + edx * 4], esi  // mat3[i * k + j] = sum

            // Increment j
            inc ebx
            jmp interno_ciclo

            fine_ciclo_interno :

        // Increment i
        inc eax
            jmp esterno_ciclo

            fine_ciclo_esterno :
    }

    // Print to screen
    {   unsigned int i, j, h;
    printf("Product matrix:\n");
    for (i = h = 0; i < m; i++)
    {
        for (j = 0; j < k; j++, h++)
            printf("%6d ", mat3[h]);
        printf("\n");
    }
    }
}

问题根源与解决建议

1. 核心问题:非法手动操作栈帧内存

原代码直接使用mov dword ptr[ebp - 4], 0手动修改栈帧内存,ebp-4属于编译器自动管理的栈区域(可能存储返回地址、栈保护金丝雀值或其他局部变量),手动写入会破坏栈结构,触发缓冲区溢出检测。

修复方案:在C代码中定义局部变量int sum,通过变量地址访问代替手动栈偏移操作,让编译器负责栈内存的安全管理。

2. 修复main函数标准合规性

标准C要求main函数返回类型为int,并在末尾返回0(表示程序正常退出)。原代码的void main()可能导致编译器对栈帧的处理异常,需修正为int main()并添加return 0;。

3. 优化结果矩阵内存分配

原代码中mat3[1024]属于固定大小的栈数组,建议改为变长数组(C99及以上支持):int mat3[m*k];,或者使用动态分配:

int *mat3 = malloc(m*k*sizeof(int));
// 使用完成后释放内存
free(mat3);

避免不必要的栈内存占用,同时防止后续修改矩阵尺寸时出现溢出。

4. 规避栈保护机制

现代编译器默认开启栈金丝雀(Stack Canary)保护,直接操作ebp附近内存会触发金丝雀值校验失败,导致程序崩溃。使用局部变量访问可完全避开该问题。

修复后的完整代码

#include <stdio.h>

int main()
{
    // 第一个矩阵的行数
    unsigned int m = 3;
    // 第一个矩阵的列数
    unsigned int n = 2;
    // 第二个矩阵的列数
    unsigned int k = 4;
    // 第一个矩阵
    short int mat1[] = { -1,-2, 4,5, 4,-2 };
    // 第二个矩阵
    short int mat2[] = { 2,0,0,0, 0,2,0,0 };
    // 结果矩阵:使用变长数组,按需分配
    int mat3[m*k];
    // 局部变量存储累加和,替代手动栈内存操作
    int sum;

    __asm {
        // 初始化行索引为0
        mov eax, 0          // i = 0

        // 外层循环:遍历结果矩阵的行
        esterno_ciclo:
        // 判断是否遍历完所有行
        cmp eax, m
            jge fine_ciclo_esterno  // 如果i >= m,退出外层循环

            // 初始化列索引为0
            mov ebx, 0          // j = 0

            // 内层循环:遍历结果矩阵的列
            interno_ciclo:
        // 判断是否遍历完所有列
        cmp ebx, k
            jge fine_ciclo_interno  // 如果j >= k,退出内层循环

            // 初始化当前单元格的累加和为0
            mov dword ptr[sum], 0  

            // 初始化第一个矩阵的列索引为0
            mov ecx, 0  // l = 0

            // 中间循环:计算矩阵元素乘积并累加
            ciclo_medio:
        // 判断是否遍历完第一个矩阵的所有列
        cmp ecx, n
            jge fine_ciclo_medio  // 如果l >= n,退出中间循环

            // 计算当前mat1元素的地址
            mov edx, eax                      // edx = i
            imul edx, n                       // edx = i * n
            add edx, ecx                      // edx = i * n + l
            movsx esi, word ptr[mat1 + edx * 2]  // esi = mat1[i * n + l]

            // 计算当前mat2元素的地址
            mov edx, ecx                      // edx = l
            imul edx, k                       // edx = l * k
            add edx, ebx                      // edx = l * k + j
            movsx edi, word ptr[mat2 + edx * 2]  // edi = mat2[l * k + j]

            // 乘积并累加到sum
            imul esi, edi                     // esi = mat1[i*n+l] * mat2[l*k+j]
            add dword ptr[sum], esi     // sum += 乘积结果

            // 递增l
            inc ecx
            jmp ciclo_medio

            fine_ciclo_medio :

        // 将累加和存入结果矩阵
        mov edx, eax                   // edx = i
            imul edx, k                    // edx = i * k
            add edx, ebx                   // edx = i * k + j
            mov esi, dword ptr[sum]  // esi = sum
            mov dword ptr[mat3 + edx * 4], esi  // mat3[i*k+j] = sum

            // 递增j
            inc ebx
            jmp interno_ciclo

            fine_ciclo_interno :

        // 递增i
        inc eax
            jmp esterno_ciclo

            fine_ciclo_esterno :
    }

    // 打印结果
    {   
        unsigned int i, j, h;
        printf("Product matrix:\n");
        for (i = h = 0; i < m; i++)
        {
            for (j = 0; j < k; j++, h++)
                printf("%6d ", mat3[h]);
            printf("\n");
        }
    }
    return 0;
}

内容的提问来源于stack exchange,提问作者TheFage

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 11:37:32