一、安装#
git clone https://github.com/google/benchmark.git
cd benchmark
cmake -DCMAKE_BUILD_TYPE=Release -DBENCHMARK_DOWNLOAD_DEPENDENCIES=ON
make -j4
sudo make install
二、编写第一个基准测试#
#include <benchmark/benchmark.h>
#include <vector>
#include <numeric>
// 测试 vector 求和的不同方式
static void BM_Sum_Loop(benchmark::State& state) {
std::vector<int> v(state.range(0));
std::iota(v.begin(), v.end(), 0);
for (auto _ : state) {
long long sum = 0;
for (int x : v) sum += x;
benchmark::DoNotOptimize(sum);
}
}
BENCHMARK(BM_Sum_Loop)->Arg(1000)->Arg(10000)->Arg(100000);
static void BM_Sum_STL(benchmark::State& state) {
std::vector<int> v(state.range(0));
std::iota(v.begin(), v.end(), 0);
for (auto _ : state) {
auto sum = std::accumulate(v.begin(), v.end(), 0LL);
benchmark::DoNotOptimize(sum);
}
}
BENCHMARK(BM_Sum_STL)->Arg(1000)->Arg(10000)->Arg(100000);
BENCHMARK_MAIN();
# 编译并运行
g++ -O2 -o bench bench.cpp -lbenchmark
./bench
# 输出示例:
# --------------------------------------------------------------
# Benchmark Time CPU Iterations
# --------------------------------------------------------------
# BM_Sum_Loop/1000 432 ns 431 ns 1623905
# BM_Sum_Loop/10000 4123 ns 4120 ns 169853
# BM_Sum_Loop/100000 41234 ns 41230 ns 16985
# BM_STL_Sum/1000 428 ns 427 ns 1638281
# BM_STL_Sum/10000 4128 ns 4125 ns 169720
# BM_STL_Sum/100000 41280 ns 41270 ns 16972
# 分析:for 循环和 std::accumulate 几乎没区别(编译器优化到一样了)
三、for (auto _ : state) 的作用#
// 这个循环框架帮做了 3 件事:
static void BM_Example(benchmark::State& state) {
// ① 初始化(只在测试开始前执行一次)
std::vector<int> data(1000000);
// ② 被测代码在循环中重复执行
for (auto _ : state) {
// benchmark 框架会:
// - 自动计算执行了多少次以达到统计显著性
// - 排除循环本身的时间开销
// - 多次运行取平均值
do_something(data);
}
// ③ 这里的代码只有在所有循环结束后执行一次
// 可以在测试结束时做结果校验
}