You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何Virtual+Indirect函数调用比单独调用更快?性能测试分析

间接调用与虚函数叠加调用的性能反常问题

测试目标

测量四种函数调用类型的性能开销,验证叠加间接调用是否会累加性能损耗:

  • Direct function call:直接调用同一DLL内的函数(可能被内联)
  • Indirect function call:通过指针调用另一DLL内的函数(未被内联)
  • virtual function call:虚函数调用
  • virtual function call + indirect function call:持有函数指针的多态类调用

为避免编译器内联优化,将代码拆分为可执行文件(exe)和动态链接库(dll)两部分。

测试代码

可执行文件代码

#include <limits.h>
#include <vector>
#include <chrono>
#include <iostream>
#include <span>

static int foo(int a, std::vector<int>& v) {
    v.push_back(a);
    if (v.size() > 2)
    {
        v.clear();
    }
    return a;
}

struct IFooable
{
    virtual int foo(int a) = 0;
};

__declspec(dllimport) int foo2(int a, std::vector<int>& v);

__declspec(dllimport) int direct_version(std::vector<int>& v);

__declspec(dllimport) int indirect_version(int (*fn)(int, void*), void* p);

__declspec(dllimport) int indirect_Interface(IFooable& f);

struct MyFoo final: public IFooable
{
    int foo(int a) override
    {
        return ::foo(a, v);
    }
    MyFoo(std::vector<int>& v) : v{ v } {}
    std::vector<int>& v;
};

struct MyFoo2 final : public IFooable
{
    using functype = int (*)(int, std::vector<int>&);
    int foo(int a) override
    {
        return f(a, v);
    }
    MyFoo2(std::vector<int>& v, functype f) : v{ v }, f{ f } {}
    std::vector<int>& v;
    functype f;
};

int main(int argc, char* argv[]) {
    std::vector<int> v;
    for (int i = 0; i < 20; i++)
    {
        foo(i, v);
    }

    std::chrono::steady_clock::time_point begin = std::chrono::steady_clock::now();

    direct_version(v);
    std::chrono::steady_clock::time_point end1 = std::chrono::steady_clock::now();

    indirect_version([](int a, void* p) {return foo(a, *reinterpret_cast<std::vector<int>*>(p)); }, reinterpret_cast<void*>(&v));
    std::chrono::steady_clock::time_point end2 = std::chrono::steady_clock::now();

    MyFoo ff{ v };
    indirect_Interface(ff);
    std::chrono::steady_clock::time_point end3 = std::chrono::steady_clock::now();
    MyFoo2 ff2{ v, foo2 };
    indirect_Interface(ff2);
    std::chrono::steady_clock::time_point end4 = std::chrono::steady_clock::now();

    std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end1 - begin).count() << "[ms] " << "Direct" << std::endl;
    std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end2 - end1).count() << "[ms] " << "Indirect" << std::endl;
    std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end3 - end2).count() << "[ms] " << "Virtual" << std::endl;
    std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end4 - end3).count() << "[ms] " << "Virtual + Indirect" << std::endl;

    double micros_count = std::chrono::duration_cast<std::chrono::milliseconds>(end2 - end1).count();
    double iterations = INT_MAX;
    std::cout << "nanoseconds per iteration = " << (micros_count / iterations) * 1000000 << '\n';
    
    return 0;
}

DLL代码

// dllmain.cpp : Defines the entry point for the DLL application.
#include "pch.h"
#include <vector>
#include <span>

BOOL APIENTRY DllMain( HMODULE hModule,
                       DWORD  ul_reason_for_call,
                       LPVOID lpReserved
                     )
{
    switch (ul_reason_for_call)
    {
    case DLL_PROCESS_ATTACH:
    case DLL_THREAD_ATTACH:
    case DLL_THREAD_DETACH:
    case DLL_PROCESS_DETACH:
        break;
    }
    return TRUE;
}

struct IFooable
{
    virtual int foo(int a) = 0;
};

static int foo(int a, std::vector<int>& v) {
    v.push_back(a);
    if (v.size() > 2)
    {
        v.clear();
    }
    return a;
}

__declspec(dllexport) int foo2(int a, std::vector<int>& v) {
    v.push_back(a);
    if (v.size() > 2)
    {
        v.clear();
    }
    return a;
}

__declspec(dllexport) int direct_version(std::vector<int>& v) {
    int i, b = 0;
    for (i = 0; i < INT_MAX; ++i) {
        b = foo(b, v);
    }
    return b;
}

__declspec(dllexport) int indirect_version(int (*fn)(int, void*), void* p) {
    int i, b = 0;

    for (i = 0; i < INT_MAX; ++i) {
        b = fn(b, p);
    }

    return b;
}

__declspec(dllexport) int indirect_Interface(IFooable& f) {
    int i, b = 0;

    for (i = 0; i < INT_MAX; ++i) {
        b = f.foo(b);
    }

    return b;
}

测试设置

  • 编译模式:Release(/O2优化)
  • 设计目标:避免缓存缺失,确保CPU能预测函数指向,排除分支预测失败或缓存缺失的影响
  • 验证:已检查汇编确认无内联优化

测试结果

初始测试结果

3058[ms] Direct
8279[ms] Indirect
9109[ms] Virtual
7340[ms] Virtual + Indirect
nanoseconds per iteration = 3.85521

关闭Buffer overflow checks后的结果

3051[ms] Direct
6182[ms] Indirect
8002[ms] Virtual
7616[ms] Virtual + Indirect
nanoseconds per iteration = 2.87872

反常现象与核心问题

虚函数调用比间接调用略慢符合预期,但Virtual+Indirect调用反而比单独虚函数调用更快,这与“叠加间接调用会累加性能损耗”的预期不符。

核心疑问:

  1. 为何Virtual+Indirect函数调用比单独虚函数调用更快?
  2. 实际场景中这类叠加间接调用的开销是否会累加?

注:测试已排除调用顺序、CPU升频/过热、调度等干扰因素,场景贴近真实的“C++虚函数API封装另一DLL中C API”的常见情况。


内容的提问来源于stack exchange,提问作者Ahmed AEK

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.14 01:14:58