为何Virtual+Indirect函数调用比单独调用更快?性能测试分析
间接调用与虚函数叠加调用的性能反常问题
测试目标
测量四种函数调用类型的性能开销,验证叠加间接调用是否会累加性能损耗:
- Direct function call:直接调用同一DLL内的函数(可能被内联)
- Indirect function call:通过指针调用另一DLL内的函数(未被内联)
- virtual function call:虚函数调用
- virtual function call + indirect function call:持有函数指针的多态类调用
为避免编译器内联优化,将代码拆分为可执行文件(exe)和动态链接库(dll)两部分。
测试代码
可执行文件代码
#include <limits.h> #include <vector> #include <chrono> #include <iostream> #include <span> static int foo(int a, std::vector<int>& v) { v.push_back(a); if (v.size() > 2) { v.clear(); } return a; } struct IFooable { virtual int foo(int a) = 0; }; __declspec(dllimport) int foo2(int a, std::vector<int>& v); __declspec(dllimport) int direct_version(std::vector<int>& v); __declspec(dllimport) int indirect_version(int (*fn)(int, void*), void* p); __declspec(dllimport) int indirect_Interface(IFooable& f); struct MyFoo final: public IFooable { int foo(int a) override { return ::foo(a, v); } MyFoo(std::vector<int>& v) : v{ v } {} std::vector<int>& v; }; struct MyFoo2 final : public IFooable { using functype = int (*)(int, std::vector<int>&); int foo(int a) override { return f(a, v); } MyFoo2(std::vector<int>& v, functype f) : v{ v }, f{ f } {} std::vector<int>& v; functype f; }; int main(int argc, char* argv[]) { std::vector<int> v; for (int i = 0; i < 20; i++) { foo(i, v); } std::chrono::steady_clock::time_point begin = std::chrono::steady_clock::now(); direct_version(v); std::chrono::steady_clock::time_point end1 = std::chrono::steady_clock::now(); indirect_version([](int a, void* p) {return foo(a, *reinterpret_cast<std::vector<int>*>(p)); }, reinterpret_cast<void*>(&v)); std::chrono::steady_clock::time_point end2 = std::chrono::steady_clock::now(); MyFoo ff{ v }; indirect_Interface(ff); std::chrono::steady_clock::time_point end3 = std::chrono::steady_clock::now(); MyFoo2 ff2{ v, foo2 }; indirect_Interface(ff2); std::chrono::steady_clock::time_point end4 = std::chrono::steady_clock::now(); std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end1 - begin).count() << "[ms] " << "Direct" << std::endl; std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end2 - end1).count() << "[ms] " << "Indirect" << std::endl; std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end3 - end2).count() << "[ms] " << "Virtual" << std::endl; std::cout << std::chrono::duration_cast<std::chrono::milliseconds>(end4 - end3).count() << "[ms] " << "Virtual + Indirect" << std::endl; double micros_count = std::chrono::duration_cast<std::chrono::milliseconds>(end2 - end1).count(); double iterations = INT_MAX; std::cout << "nanoseconds per iteration = " << (micros_count / iterations) * 1000000 << '\n'; return 0; }
DLL代码
// dllmain.cpp : Defines the entry point for the DLL application. #include "pch.h" #include <vector> #include <span> BOOL APIENTRY DllMain( HMODULE hModule, DWORD ul_reason_for_call, LPVOID lpReserved ) { switch (ul_reason_for_call) { case DLL_PROCESS_ATTACH: case DLL_THREAD_ATTACH: case DLL_THREAD_DETACH: case DLL_PROCESS_DETACH: break; } return TRUE; } struct IFooable { virtual int foo(int a) = 0; }; static int foo(int a, std::vector<int>& v) { v.push_back(a); if (v.size() > 2) { v.clear(); } return a; } __declspec(dllexport) int foo2(int a, std::vector<int>& v) { v.push_back(a); if (v.size() > 2) { v.clear(); } return a; } __declspec(dllexport) int direct_version(std::vector<int>& v) { int i, b = 0; for (i = 0; i < INT_MAX; ++i) { b = foo(b, v); } return b; } __declspec(dllexport) int indirect_version(int (*fn)(int, void*), void* p) { int i, b = 0; for (i = 0; i < INT_MAX; ++i) { b = fn(b, p); } return b; } __declspec(dllexport) int indirect_Interface(IFooable& f) { int i, b = 0; for (i = 0; i < INT_MAX; ++i) { b = f.foo(b); } return b; }
测试设置
- 编译模式:Release(/O2优化)
- 设计目标:避免缓存缺失,确保CPU能预测函数指向,排除分支预测失败或缓存缺失的影响
- 验证:已检查汇编确认无内联优化
测试结果
初始测试结果
3058[ms] Direct 8279[ms] Indirect 9109[ms] Virtual 7340[ms] Virtual + Indirect nanoseconds per iteration = 3.85521
关闭Buffer overflow checks后的结果
3051[ms] Direct 6182[ms] Indirect 8002[ms] Virtual 7616[ms] Virtual + Indirect nanoseconds per iteration = 2.87872
反常现象与核心问题
虚函数调用比间接调用略慢符合预期,但Virtual+Indirect调用反而比单独虚函数调用更快,这与“叠加间接调用会累加性能损耗”的预期不符。
核心疑问:
- 为何Virtual+Indirect函数调用比单独虚函数调用更快?
- 实际场景中这类叠加间接调用的开销是否会累加?
注:测试已排除调用顺序、CPU升频/过热、调度等干扰因素,场景贴近真实的“C++虚函数API封装另一DLL中C API”的常见情况。
内容的提问来源于stack exchange,提问作者Ahmed AEK
相关产品推荐
相关产品推荐

