You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

C++同列数多行向量高效实现优化:消除重复代码

实现高效无重复代码的多行向量(C++)

你的问题非常典型——当需要维护多组同构的指针数组时,重复编写分配/重分配/释放代码不仅繁琐,还容易引入人为错误。下面给出两种符合DRY原则且完全不牺牲性能的方案,都避免了运行时多态的开销:

方案一:用std::tuple+编译期遍历(现代C++风格)

利用C++11及以后的变参模板和std::index_sequence,我们可以把所有行指针打包进一个std::tuple,然后通过模板函数批量处理所有元素。这种方式类型安全,编译期展开代码,性能和手动编写完全一致。

修改后的代码示例

首先调整BaseVector(保持原有核心逻辑,确保模板函数可访问):

class BaseVector {
protected:
    int64_t _capacity;
    int64_t _nColumns;

protected:
    template<typename taItem>
    void Allocate(taItem*& p, const int64_t nItems) {
        p = static_cast<taItem*>(MemPool::Instance().Acquire(sizeof(taItem)*nItems));
        if (p == nullptr) {
            __debugbreak();
        }
    }

    template<typename taItem>
    void Reallocate(taItem*& p, const int64_t newCap) {
        taItem* np;
        Allocate(np, newCap);
        Utils::AlignedNocachingCopy(np, p, _nColumns * sizeof(taItem));
        MemPool::Instance().Release(p, _capacity * sizeof(taItem));
        p = np;
    }

    template<typename taItem>
    void Release(taItem*& p, const int64_t capacity) {
        if (p != nullptr) {
            MemPool::Instance().Release(p, capacity * sizeof(taItem));
            p = nullptr;
        }
    }

public:
    explicit BaseVector(const int64_t initCap) : _capacity(initCap), _nColumns(0) {}
    void Clear() { _nColumns = 0; }
    int64_t Size() const { return _nColumns; }
};

然后实现DerivedVector,用tuple管理所有行指针:

#include <tuple>
#include <utility> // 用于std::make_index_sequence

class DerivedVector : public BaseVector {
    // 把所有行指针打包进tuple,新增行只需在这里添加类型即可
    using RowTuple = std::tuple<__m256d*, __m256i*, uint64_t*, uint8_t*>;
    RowTuple _rows;

    // 编译期遍历tuple,执行分配
    template<std::size_t... Is>
    void AllocateAll(int64_t nColumns, std::index_sequence<Is...>) {
        (Allocate(std::get<Is>(_rows), nColumns), ...);
    }

    // 编译期遍历tuple,执行重分配
    template<std::size_t... Is>
    void ReallocateAll(int64_t newCap, std::index_sequence<Is...>) {
        (Reallocate(std::get<Is>(_rows), newCap), ...);
    }

    // 编译期遍历tuple,执行释放
    template<std::size_t... Is>
    void ReleaseAll(std::index_sequence<Is...>) {
        (Release(std::get<Is>(_rows), _capacity), ...);
    }

public:
    explicit DerivedVector(const int64_t nColumns) : BaseVector(nColumns) {
        // 生成对应tuple大小的索引序列,触发编译期遍历
        AllocateAll(nColumns, std::make_index_sequence<std::tuple_size_v<RowTuple>>{});
    }

    void IncSize() {
        if(_nColumns >= _capacity) {
            const int64_t newCap = _capacity + (_capacity >> 1) + 1;
            ReallocateAll(newCap, std::make_index_sequence<std::tuple_size_v<RowTuple>>{});
            _capacity = newCap;
        }
        _nColumns++;
    }

    // 访问指定行的指针(比如获取RowA)
    __m256d* GetRowA() { return std::get<0>(_rows); }
    const __m256d* GetRowA() const { return std::get<0>(_rows); }
    // 其他行的访问函数同理,或者直接提供std::get的包装

    ~DerivedVector() {
        ReleaseAll(std::make_index_sequence<std::tuple_size_v<RowTuple>>{});
    }
};

优势

  • 类型安全:编译期就能检查所有行的类型匹配
  • 扩展性强:新增一行只需在RowTuple中添加对应指针类型,无需修改其他逻辑
  • 零运行时开销:模板展开后的代码和手动编写的完全一致,没有额外调用或多态开销

方案二:用宏生成重复代码(轻量简洁)

如果你更倾向于简洁的代码风格,宏可以高效生成重复的声明和调用代码,同样是编译期处理,性能不受影响。

修改后的代码示例

class BaseVector {
    // 保持原有BaseVector代码不变...
};

// 定义宏来生成行指针声明、分配、重分配、释放代码
#define DECLARE_ROW(type, suffix) type* _pRow##suffix;
#define ALLOCATE_ROW(suffix) Allocate(_pRow##suffix, nColumns);
#define REALLOCATE_ROW(suffix) Reallocate(_pRow##suffix, newCap);
#define RELEASE_ROW(suffix) Release(_pRow##suffix, _capacity);

class DerivedVector : public BaseVector {
    // 声明所有行,新增行只需添加一行DECLARE_ROW
    DECLARE_ROW(__m256d, A)
    DECLARE_ROW(__m256i, B)
    DECLARE_ROW(uint64_t, C)
    DECLARE_ROW(uint8_t, D)
    // ... 新增行继续添加DECLARE_ROW即可

public:
    explicit DerivedVector(const int64_t nColumns) : BaseVector(nColumns) {
        // 批量分配所有行
        ALLOCATE_ROW(A)
        ALLOCATE_ROW(B)
        ALLOCATE_ROW(C)
        ALLOCATE_ROW(D)
        // ... 新增行对应添加ALLOCATE_ROW
    }

    void IncSize() {
        if(_nColumns >= _capacity) {
            const int64_t newCap = _capacity + (_capacity >> 1) + 1;
            // 批量重分配所有行
            REALLOCATE_ROW(A)
            REALLOCATE_ROW(B)
            REALLOCATE_ROW(C)
            REALLOCATE_ROW(D)
            // ... 新增行对应添加REALLOCATE_ROW
            _capacity = newCap;
        }
        _nColumns++;
    }

    // 直接访问行指针,和原来的方式完全一致
    __m256d* GetRowA() { return _pRowA; }
    const __m256d* GetRowA() const { return _pRowA; }

    ~DerivedVector() {
        // 批量释放所有行
        RELEASE_ROW(A)
        RELEASE_ROW(B)
        RELEASE_ROW(C)
        RELEASE_ROW(D)
        // ... 新增行对应添加RELEASE_ROW
    }
};

// 可选:undef宏避免污染全局命名空间
#undef DECLARE_ROW
#undef ALLOCATE_ROW
#undef REALLOCATE_ROW
#undef RELEASE_ROW

优势

  • 代码极简:重复代码被压缩成宏调用,可读性高
  • 访问性能最优:直接访问成员指针,和你原来的实现完全一致
  • 学习成本低:无需掌握复杂的模板元编程知识

性能说明

两种方案都是编译期生成代码,最终的二进制结果和你手动编写30次Allocate/Reallocate完全相同,不会引入任何运行时开销。单元格访问时直接通过指针操作,和原有实现的性能一致,完全满足你的性能要求。

内容的提问来源于stack exchange,提问作者Serge Rogatch

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.12 04:22:32