You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Python CFFI生成CSV处理共享对象供C调用时链接错误排查

链接错误排查及完整CSV处理库实现

链接错误排查步骤

  • 检查生成的共享库文件名:执行python3 csv_processor.py后,当前目录应生成_csv_processor.so(对应代码里set_source的第一个参数)。ld的-lxxx规则是查找libxxx.so,所以直接用-l_csv_processor会找不到,要么用-l:_csv_processor.so(冒号指定完整文件名),要么把库重命名为libcsv_processor.so再用-lcsv_processor。
  • 确认生成步骤:CFFI的out-of-line模式需要执行脚本编译出共享库,直接运行脚本如果没报错,会同时生成.so和.h文件;如果只生成了.c文件,需要执行python3 csv_processor.py build_ext --inplace完成编译。
  • 验证库路径:-L.指定当前目录为库搜索路径,确保.so文件确实在当前目录下。

完整实现代码

1. Python CFFI 核心实现(csv_processor.py)

from cffi import FFI

ffibuilder = FFI()

# 定义对外暴露的C接口
ffibuilder.cdef("""
    typedef struct {
        char** rows;
        int row_count;
        char** headers;
        int header_count;
    } CSVResult;

    CSVResult process_csv(const char* filepath, const char* selected_columns, const char* filter_condition);
    void free_csv_result(CSVResult result);
""")

# 声明C依赖头文件
ffibuilder.set_source("_csv_processor", """
    #include <stdio.h>
    #include <stdlib.h>
    #include <string.h>
""", source_extension='.c')

# 处理CSV的核心逻辑
@ffibuilder.def_extern()
def process_csv(filepath, selected_columns, filter_condition):
    import csv
    from collections import defaultdict

    # 初始化返回结构体
    result = ffibuilder.new("CSVResult")
    result.rows = ffibuilder.NULL
    result.row_count = 0
    result.headers = ffibuilder.NULL
    result.header_count = 0

    try:
        # 解析参数为Python字符串
        selected_cols = selected_columns.decode('utf-8').split(',') if selected_columns else []
        filters = defaultdict(list)
        
        # 解析过滤条件(支持=、>、<操作)
        if filter_condition:
            for cond in filter_condition.decode('utf-8').split(','):
                if '=' in cond:
                    key, val = cond.split('=', 1)
                    filters[key].append(('eq', val))
                elif '>' in cond:
                    key, val = cond.split('>', 1)
                    filters[key].append(('gt', val))
                elif '<' in cond:
                    key, val = cond.split('<', 1)
                    filters[key].append(('lt', val))

        # 读取并处理CSV
        with open(filepath.decode('utf-8'), 'r', newline='', encoding='utf-8') as f:
            reader = csv.DictReader(f)
            headers = reader.fieldnames
            output_headers = selected_cols if selected_cols else headers
            
            # 分配表头内存
            result.header_count = len(output_headers)
            result.headers = ffibuilder.cast("char**", ffibuilder.new("char[]", result.header_count * ffibuilder.sizeof("char*")))
            for i, header in enumerate(output_headers):
                result.headers[i] = ffibuilder.new("char[]", header.encode('utf-8'))

            # 过滤并提取行数据
            filtered_rows = []
            for row in reader:
                match = True
                # 校验所有过滤条件
                for key, conds in filters.items():
                    if key not in row:
                        match = False
                        break
                    row_val = row[key]
                    for op, val in conds:
                        if op == 'eq' and row_val != val:
                            match = False
                            break
                        elif op == 'gt':
                            try:
                                if float(row_val) <= float(val):
                                    match = False
                                    break
                            except ValueError:
                                match = False
                                break
                        elif op == 'lt':
                            try:
                                if float(row_val) >= float(val):
                                    match = False
                                    break
                            except ValueError:
                                match = False
                                break
                    if not match:
                        break
                if match:
                    output_row = [row[col] if col in row else '' for col in output_headers]
                    filtered_rows.append(output_row)

            # 分配行数据内存
            result.row_count = len(filtered_rows)
            if result.row_count > 0:
                result.rows = ffibuilder.cast("char**", ffibuilder.new("char[]", result.row_count * ffibuilder.sizeof("char*")))
                for i, row in enumerate(filtered_rows):
                    row_str = ','.join(row).encode('utf-8')
                    result.rows[i] = ffibuilder.new("char[]", row_str)

    except Exception:
        # 出错时释放已分配内存
        free_csv_result(result[0])
        result.rows = ffibuilder.NULL
        result.row_count = 0
        result.headers = ffibuilder.NULL
        result.header_count = 0

    return result[0]

# 内存释放函数,避免泄漏
@ffibuilder.def_extern()
def free_csv_result(result):
    if result.headers != ffibuilder.NULL:
        for i in range(result.header_count):
            if result.headers[i] != ffibuilder.NULL:
                ffibuilder.free(result.headers[i])
        ffibuilder.free(result.headers)
        result.headers = ffibuilder.NULL
    if result.rows != ffibuilder.NULL:
        for i in range(result.row_count):
            if result.rows[i] != ffibuilder.NULL:
                ffibuilder.free(result.rows[i])
        ffibuilder.free(result.rows)
        result.rows = ffibuilder.NULL
    result.row_count = 0
    result.header_count = 0

if __name__ == "__main__":
    ffibuilder.compile(verbose=True)

2. C调用示例代码(programa_c.c)

#include <stdio.h>
#include <stdlib.h>

// 引入CFFI生成的头文件
#include "_csv_processor.h"

int main() {
    // 参数:CSV文件路径、选中列(逗号分隔)、过滤条件(逗号分隔的键值/比较表达式)
    CSVResult result = process_csv("data.csv", "name,age", "age>30");

    if (result.header_count == 0 || result.row_count == 0) {
        printf("无返回数据或处理出错\n");
        free_csv_result(result);
        return 1;
    }

    // 打印表头
    printf("表头:");
    for (int i = 0; i < result.header_count; i++) {
        printf("%s ", result.headers[i]);
    }
    printf("\n");

    // 打印行数据
    printf("数据行:\n");
    for (int i = 0; i < result.row_count; i++) {
        printf("%s\n", result.rows[i]);
    }

    // 必须释放内存
    free_csv_result(result);

    return 0;
}

编译与运行步骤

  1. 生成共享库:执行python3 csv_processor.py,会生成_csv_processor.so和_csv_processor.h。
  2. 编译C程序:使用以下命令(直接指定库文件名)
    gcc -o programa_c programa_c.c -L. -l:_csv_processor.so -lpython3.10
    
    或者重命名库后编译:
    mv _csv_processor.so libcsv_processor.so
    gcc -o programa_c programa_c.c -L. -lcsv_processor -lpython3.10
    
  3. 测试运行:准备data.csv测试文件(例如包含name,age,gender列),执行./programa_c查看输出。

内容的提问来源于stack exchange,提问作者João Viitor

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.21 12:34:55