You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

PyInstaller打包可执行文件时如何规避完整导入Numpy?

完全弃用Numpy,大幅减小打包体积的方案

当然可以完全摆脱Numpy的依赖!你只用了Numpy来处理动态大小的数组转换和创建,这部分完全可以用Cython结合C标准库的内存操作来实现,不需要整个Numpy包。下面是具体的修改步骤和代码示例:

核心思路

我们用C的malloc/free手动分配内存,把Python元组的数据复制到C数组中,再包装成Cython的memoryview(你已经熟悉这个工具),完全替代Numpy的数组转换和创建功能。同时要注意内存泄漏问题,用try...finally确保所有分配的内存都能被释放。

具体修改步骤

1. 替换Numpy导入为C标准库

首先移除所有Numpy相关的导入,换成C标准库的内存操作函数:

cimport cython
from libc.stdlib cimport malloc, free
from libc.string cimport memcpy

2. 实现辅助函数处理数组转换和操作

这些函数用来替代Numpy的array、empty、copy等功能:

# 一维元组转int memoryview(C顺序)
cdef int[:] tuple_to_int_1d(tuple t):
    cdef Py_ssize_t n = len(t)
    cdef int* arr = <int*>malloc(n * sizeof(int))
    if not arr:
        raise MemoryError("Failed to allocate memory")
    try:
        for i in range(n):
            arr[i] = t[i]
        return <int[:n]>arr
    except:
        free(arr)
        raise

# 二维元组转Fortran顺序的int memoryview(对应你原来的order='F')
cdef int[::1, :] tuple_to_int_2d_f(tuple t):
    cdef Py_ssize_t rows = len(t)
    if rows == 0:
        return <int[:0,:0:1]>NULL
    cdef Py_ssize_t cols = len(t[0])
    cdef int* arr = <int*>malloc(rows * cols * sizeof(int))
    if not arr:
        raise MemoryError("Failed to allocate memory")
    try:
        # Fortran顺序是列优先,所以先遍历列再遍历行
        for j in range(cols):
            for i in range(rows):
                arr[j*rows + i] = t[i][j]
        return <int[::rows, :cols]>arr
    except:
        free(arr)
        raise

# 创建空的一维int memoryview
cdef int[:] create_empty_int_1d(Py_ssize_t n):
    cdef int* arr = <int*>malloc(n * sizeof(int))
    if not arr:
        raise MemoryError("Failed to allocate memory")
    return <int[:n]>arr

# 创建空的二维int memoryview(C顺序)
cdef int[:, :] create_empty_int_2d(Py_ssize_t rows, Py_ssize_t cols):
    cdef int* arr = <int*>malloc(rows * cols * sizeof(int))
    if not arr:
        raise MemoryError("Failed to allocate memory")
    return <int[:rows,:cols]>arr

# 复制一维数组
cdef int[:] copy_int_1d(int[:] src):
    cdef Py_ssize_t n = src.shape[0]
    cdef int[:] dst = create_empty_int_1d(n)
    for i in range(n):
        dst[i] = src[i]
    return dst

# 复制一维数组到另一个数组(用于替换arr[j,:] = arr[jj,:])
cdef void copy_int_1d_to(int[:] dst, int[:] src):
    cdef Py_ssize_t n = src.shape[0]
    if dst.shape[0] != n:
        raise ValueError("Destination and source must match in length")
    for i in range(n):
        dst[i] = src[i]

# 转置并复制二维数组(替换arr_.T.copy())
cdef int[:, :] transpose_and_copy_int_2d(int[:, :] src):
    cdef Py_ssize_t rows = src.shape[0]
    cdef Py_ssize_t cols = src.shape[1]
    cdef int[:, :] dst = create_empty_int_2d(cols, rows)
    for i in range(rows):
        for j in range(cols):
            dst[j,i] = src[i,j]
    return dst

3. 修改Screen函数,替换所有Numpy调用

把原来用Numpy的地方换成上面的辅助函数,同时用try...finally管理内存:

@cython.wraparound(False)
@cython.cdivision(True)
def Screen(tuple families, tuple solution_vector, Py_ssize_t solution_code, Py_ssize_t no_triplets):
    cdef int [::1, :] input_data = NULL
    cdef int [:] solution = NULL
    cdef int [:] combinations = NULL
    cdef int [:, :] Augmented = NULL
    cdef int [:, :] arr, arr_, new_arr
    cdef int [:] temp, sol, ratios, sol_indices
    cdef int x_max, y_max, i, j, ii, det, g, c_
    cdef int N, M, I, J, K, CombinedAminoAcids, indexing, is_negative, half_mark, mcd
    cdef list output = []
    
    try:
        # 替换numpy.array转换输入
        input_data = tuple_to_int_2d_f(families)
        solution = tuple_to_int_1d(solution_vector)
        
        combinations = create_empty_int_1d(no_triplets)
        N = solution.shape[0]
        M = input_data.shape[0]
        Augmented = create_empty_int_2d(no_triplets+1, N)
        
        # 复制solution到Augmented最后一行(替换原来的直接赋值)
        for i in range(N):
            Augmented[no_triplets, i] = solution[i]
        
        # 初始化combinations数组
        for I in range(no_triplets):
            combinations[I] = M -1 - I
        combinations[no_triplets - 1] += 1
        
        indexing = 0
        half_mark = no_triplets // 2 + 1
        
        while combinations[0] > no_triplets - 1:
            K = no_triplets - 1
            for J in range(no_triplets - 1, -1, -1):
                if combinations[J] > no_triplets - 1 - J:
                    K = J
                    break
            combinations[K] -= 1
            for J in range(1, no_triplets - K):
                combinations[K + J] = combinations[K] - J
            
            CombinedAminoAcids = 0
            for I in range(no_triplets):
                CombinedAminoAcids |= input_data[combinations[I], 0]
            
            if CombinedAminoAcids == solution_code:
                # 复制vectors到Augmented前no_triplets行
                for I in range(no_triplets):
                    for idx in range(N):
                        Augmented[I, idx] = input_data[combinations[I], 3 + idx]
                
                indexing += 1
                ratios = C_Gauss(Augmented)
                sol_indices = copy_int_1d(ratios)
                
                is_negative = 0
                for I in range(no_triplets):
                    if ratios[I] == 0:
                        break
                    elif ratios[I] < 0:
                        is_negative += 1
                    ratios[I] *= input_data[combinations[I], 2]
                    sol_indices[I] = input_data[combinations[I], 1]
                else:
                    mcd = GCD(ratios)
                    if is_negative >= half_mark:
                        mcd *= -1
                    for I in range(no_triplets):
                        ratios[I] /= mcd
                    output.append( (tuple(sol_indices), tuple(ratios)) )
        
        return output
    
    finally:
        # 确保所有分配的内存都被释放,避免泄漏
        if input_data != NULL:
            free(&input_data[0,0])
        if solution != NULL:
            free(&solution[0])
        if combinations != NULL:
            free(&combinations[0])
        if Augmented != NULL:
            free(&Augmented[0,0])

4. 修改C_Gauss函数中的数组操作

比如原来的temp = arr[j, :].copy()和交换行的逻辑,替换成我们的辅助函数:

@cython.wraparound(False)
@cython.cdivision(True)
cdef int [:] C_Gauss(int [:, :] Arr):
    cdef int [:, :] arr, arr_, new_arr
    cdef int x_max, y_max, i, j, ii, det, g, c_
    cdef int [:] temp, sol
    arr_ = Arr
    arr = transpose_and_copy_int_2d(arr_)  # 替换arr_.T.copy()
    x_max, y_max = arr.shape[0], arr.shape[1]
    
    # 替换arr_[:y_max-1, 0].copy()和sol[:] = 0
    sol = create_empty_int_1d(y_max-1)
    for i in range(y_max-1):
        sol[i] = 0
    
    for j in range (y_max-1):
        if arr[j,j] == 0:
            for jj in range(j+1, x_max):
                if arr[jj, j] != 0:
                    # 交换两行:替换temp = arr[j, :].copy()等操作
                    temp = create_empty_int_1d(y_max)
                    copy_int_1d_to(temp, arr[j, :])
                    copy_int_1d_to(arr[j, :], arr[jj, :])
                    copy_int_1d_to(arr[jj, :], temp)
                    free(&temp[0])  # 用完释放temp内存
                    break
            else:
                break
        
        for i in range(j+1, x_max):
            if arr[i, j] != 0:
                for jj in range(y_max-1, j-1, -1):
                    arr[i, jj] = arr[j,j]*arr [i, jj]- arr[i,j]*arr[j,jj]
                g = GCD(arr[i, :])
                if g != 1:
                    for jj in range(y_max-1, j, -1):
                        arr [i , jj] = arr [i , jj]//g
        else:
            for jj in range(y_max-1, x_max):
                if arr[jj, y_max-1] != 0:
                    break
            else:
                new_arr = arr[:y_max-1, :]
                det = 1
                for j in range (y_max-2, 0, -1):
                    det *= new_arr[j,j]
                for i in range(j):
                    if new_arr[i, j] != 0:
                        c_ = new_arr[i,j]
                        for jj in range(i, y_max):
                            new_arr[i, jj] = new_arr[i,jj]*new_arr[j, j] - new_arr[j,jj]*c_
                        g = GCD(new_arr[i, :])
                        for jj in range(i, y_max):
                            new_arr [i, jj] = new_arr [i, jj]//g
                for j in range(y_max-1):
                    new_arr[j, y_max-1] *= det
                    new_arr[j, y_max-1] = new_arr[j, y_max-1]//new_arr[j, j]
                sol = new_arr[:, y_max-1]
    
    # 释放arr的内存
    free(&arr[0,0])
    return sol

5. 编译配置调整

修改你的setup.py,移除Numpy相关的依赖配置,比如:

from setuptools import setup, Extension
from Cython.Build import cythonize

ext_modules = [
    Extension(
        "your_module_name",
        ["your_module_name.pyx"],
        extra_compile_args=["-O3"],  # 保持优化
    )
]

setup(
    name="your_module_name",
    ext_modules=cythonize(ext_modules),
)

验证和注意事项

  1. 功能验证:修改后一定要仔细测试,确保线性代数计算的结果和原来一致,特别是数组的顺序(Fortran/C顺序)和索引是否正确。
  2. 内存泄漏检查:可以用工具(比如valgrind)检查是否有内存泄漏,确保所有malloc的内存都被free。
  3. 错误处理:辅助函数中加入了基本的错误处理(比如内存分配失败、行长度不一致),可以根据需要扩展。

这样修改后,你的Cython模块就完全不依赖Numpy了,打包后的可执行文件体积会回到10MB左右,同时保持原来的计算效率。

内容的提问来源于stack exchange,提问作者Daniele

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.08 10:37:34