diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml new file mode 100644 index 0000000..c735fd4 --- /dev/null +++ b/.github/workflows/benchmark.yml @@ -0,0 +1,60 @@ +name: Benchmark + +on: + push: + branches: [main] + pull_request: {} + workflow_dispatch: {} + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.job }}-${{ github.ref }} + cancel-in-progress: true + +defaults: + run: + shell: bash -e -l {0} + +jobs: + benchmarks: + name: Linux - ${{ matrix.sys.name }} + runs-on: namespace-profile-benchmark;container.privileged=true;container.host-pid-namespace=true + strategy: + matrix: + sys: + - { name: "sse4.2", flags: "-march=x86-64-v2" } + - { name: "avx2", flags: "-march=x86-64-v3" } + steps: + - name: Checkout code + uses: actions/checkout@v6 + - name: Set conda environment + uses: mamba-org/setup-micromamba@v2 + with: + environment-name: myenv + create-args: >- + cmake + ninja + benchmark + xsimd=14.3.0 + init-shell: bash + cache-downloads: true + - name: Build benchmarks + run: | + cmake -G Ninja -B build/ \ + -D CMAKE_BUILD_TYPE="RelWithDebInfo" \ + -D CMAKE_CXX_FLAGS="${{ matrix.sys.flags }}" \ + -D BUILD_BENCHMARK=ON + cmake --build build --target benchmark_xsimd_algorithm --parallel 8 + - name: Run benchmarks + run: | + ./build/benchmark/benchmark_xsimd_algorithm \ + --benchmark_repetitions=3 \ + --benchmark_out=benchmark-${{ matrix.sys.name }}.json \ + --benchmark_out_format=json + - name: Upload results + uses: actions/upload-artifact@v4 + with: + name: benchmark-${{ github.event.pull_request.head.sha || github.sha }}-${{ matrix.sys.name }} + path: benchmark-${{ matrix.sys.name }}.json diff --git a/benchmark/CMakeLists.txt b/benchmark/CMakeLists.txt index e214e36..ad27859 100644 --- a/benchmark/CMakeLists.txt +++ b/benchmark/CMakeLists.txt @@ -16,12 +16,25 @@ endif () find_package(benchmark REQUIRED) -set(XSIMD_ALGORITHM_BENCHMARKS +# Regular benchmark of algorithm implemented here +add_executable( + benchmark_xsimd_algorithm main.cpp - bench_math.cpp bench_map.cpp ) +target_link_libraries( + benchmark_xsimd_algorithm + PRIVATE xsimd-algorithm xsimd::test-utils benchmark::benchmark +) -add_executable(benchmark_xsimd_algorithm ${XSIMD_ALGORITHM_BENCHMARKS}) -target_link_libraries(benchmark_xsimd_algorithm - PRIVATE xsimd-algorithm xsimd::test-utils benchmark::benchmark) +# Benchmark actually targeting xsimd functionalities (hot loop in L1 cache, +# no header and trailers). Should probably be moved to xsimd itself. +add_executable( + benchmark_xsimd + main.cpp + bench_math.cpp +) +target_link_libraries( + benchmark_xsimd + PRIVATE xsimd-algorithm xsimd::test-utils benchmark::benchmark +) diff --git a/benchmark/bench_map.cpp b/benchmark/bench_map.cpp index 562c53a..44cea96 100644 --- a/benchmark/bench_map.cpp +++ b/benchmark/bench_map.cpp @@ -145,13 +145,9 @@ namespace bool const registered = [] { register_benches>(); - register_benches>(); - register_benches>(); register_benches>(); register_benches>(); - register_benches>(); register_binary_benches>(); - register_binary_benches>(); register_binary_benches(); return true; }(); diff --git a/benchmark/bench_utils.hpp b/benchmark/bench_utils.hpp index f7b3686..586eb41 100644 --- a/benchmark/bench_utils.hpp +++ b/benchmark/bench_utils.hpp @@ -22,14 +22,9 @@ namespace xsimd::bench { constexpr auto batch_size = static_cast(xsimd::batch::size); - auto sizes = std::vector {}; - for (std::int64_t size : { 64, 1024, 65536, 1 << 21 }) - { - auto const whole = size - (size % batch_size); - sizes.push_back(whole); - sizes.push_back(whole + batch_size / 2 + 1); - } - return sizes; + return std::vector { + 64, 67, 1024, 1027, 65536, 2097152 + }; } template