Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
Show all changes
27 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Prev Previous commit
Next Next commit
Split benchmarks
  • Loading branch information
AntoinePrv committed Sep 16, 2026
commit 2afa6a51fcfe37a90d5bd81599e1db122dd600d8
1 change: 1 addition & 0 deletions benchmark/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@ find_package(benchmark REQUIRED)
set(XSIMD_ALGORITHM_BENCHMARKS
main.cpp
bench_math.cpp
bench_map_unary.cpp
)

add_executable(benchmark_xsimd_algorithm ${XSIMD_ALGORITHM_BENCHMARKS})
Expand Down
54 changes: 54 additions & 0 deletions benchmark/bench_map_unary.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
/****************************************************************************
* Copyright (c) xsimd-algorithm contributors *
* *
* Distributed under the terms of the BSD 3-Clause License. *
* *
* The full license is in the file LICENSE, distributed with this software. *
****************************************************************************/

#include <cstddef>

#include <benchmark/benchmark.h>

#include "map_unary_utils.hpp"
#include "xsimd_test_utils/map_unary_data.hpp"
#include "xsimd_test_utils/utils.hpp"

namespace
{
using xsimd::bench::bench_map_unary;
using xsimd::bench::bench_scalar;
using xsimd::bench::bench_transform;
using xsimd::bench::register_bench;
using xsimd::builder::alignment;

template <typename Op>
void register_benches()
{
using input_t = typename Op::input_t;
using arch = xsimd::default_arch;
using aligned_alloc = typename xsimd::test::aligned_vector<input_t, arch>::allocator_type;
using unaligned_alloc = typename xsimd::test::unaligned_vector<input_t, arch>::allocator_type;

register_bench<Op, arch>("aligned/scalar", bench_scalar<Op, aligned_alloc>);
register_bench<Op, arch>("aligned/simd/map:header+trailer", bench_map_unary<Op, aligned_alloc, arch>);
register_bench<Op, arch>("aligned/simd/map:trailer", bench_map_unary<Op, aligned_alloc, arch, alignment { .start_aligned = true }>);
register_bench<Op, arch>("aligned/simd/transform:header+trailer", bench_transform<Op, aligned_alloc, arch>);

register_bench<Op, arch>("unaligned/scalar", bench_scalar<Op, unaligned_alloc>);
register_bench<Op, arch>("unaligned/simd/map:header+trailer", bench_map_unary<Op, unaligned_alloc, arch>);
register_bench<Op, arch>("unaligned/simd/map:trailer", bench_map_unary<Op, unaligned_alloc, arch, alignment { .start_aligned = true }>);
register_bench<Op, arch>("unaligned/simd/transform:header+trailer", bench_transform<Op, unaligned_alloc, arch>);
}

bool const registered = []
{
register_benches<xsimd::test::sqrt_op<float>>();
register_benches<xsimd::test::sqrt_op<double>>();
register_benches<xsimd::test::abs_op<float>>();
register_benches<xsimd::test::abs_op<double>>();
register_benches<xsimd::test::exp_op<float>>();
register_benches<xsimd::test::exp_op<double>>();
return true;
}();
}
100 changes: 24 additions & 76 deletions benchmark/bench_math.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -7,97 +7,43 @@
****************************************************************************/

#include <cstddef>
#include <format>
#include <string_view>

#include <benchmark/benchmark.h>

#include "bench_utils.hpp"
#include "map_unary_utils.hpp"
#include "xsimd_test_utils/map_unary_data.hpp"
#include "xsimd_test_utils/utils.hpp"

using xsimd::builder::alignment;

namespace
{
template <typename Op, typename Alloc, typename Apply>
void bench_unary(benchmark::State& state, Apply apply)
{
using input_t = typename Op::input_t;

auto const size = static_cast<std::size_t>(state.range(0));
auto [input, output] = Op::template make_input_output<Alloc>(size);

for (auto _ : state)
{
apply(xsimd::test::as_span(input), xsimd::test::as_span(output));
benchmark::DoNotOptimize(output.data());
benchmark::ClobberMemory();
}

state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * size));
state.SetBytesProcessed(
static_cast<std::int64_t>(state.iterations() * size * 2 * sizeof(input_t)));
}

template <typename Op, typename Alloc, typename Arch, alignment aligned = alignment {}>
void bench_map_unary(benchmark::State& state)
{
bench_unary<Op, Alloc>(
state,
[](auto in, auto out)
{ Op::template range_apply_map_unary<aligned, Arch>(in, out); });
}

template <typename Op, typename Alloc, typename Arch>
void bench_transform(benchmark::State& state)
{
bench_unary<Op, Alloc>(
state,
[](auto in, auto out)
{ Op::template range_apply_transform<Arch>(in, out); });
}

template <typename Op, typename Alloc>
void bench_scalar(benchmark::State& state)
{
bench_unary<Op, Alloc>(state, [](auto in, auto out)
{ Op::range_apply_scalar(in, out); });
}

template <typename Op, typename Arch, typename Bench>
void register_bench(std::string_view variant, Bench bench_fn)
{
using input_t = typename Op::input_t;

auto* bench = benchmark::RegisterBenchmark(
std::format("{}/{}/{}/{}", Arch::name(), Op::name, xsimd::bench::type_name<input_t>(), variant),
bench_fn);
for (auto const size : xsimd::bench::bench_sizes<input_t>())
{
bench->Arg(size);
}
}

using xsimd::bench::bench_map_unary;
using xsimd::bench::bench_scalar;
using xsimd::bench::register_bench;
using xsimd::builder::alignment;

/// Register math benchmarks.
///
/// To avoid an explosion of benchmarks, we only add a simple aligned benchmark.
/// This will let us know the performance of the xsimd wrappers.
/// See bench_map_unary for benchmarks on the different flavor of mapping, alignment,
/// headers and trailers.
///
/// This benchmark aims to test raw the performance of intrinsic, unrelated to
/// how they are iterated on (alignment, memory etc). To do so, they aim to stay
/// in L1 cache.
template <typename Op>
void register_benches()
{
using input_t = typename Op::input_t;
using arch = xsimd::default_arch;
using aligned_alloc = typename xsimd::test::aligned_vector<input_t, arch>::allocator_type;
using unaligned_alloc = typename xsimd::test::unaligned_vector<input_t, arch>::allocator_type;

// To avoid an explosion of benchmarks, we probagly want to only benchmark aligned for
// math ops, and benchmark map_unary/transform setups (alignment...) separately on a
// few ops.
register_bench<Op, arch>("scalar/aligned", bench_scalar<Op, aligned_alloc>);
register_bench<Op, arch>("simd-map/aligned", bench_map_unary<Op, aligned_alloc, arch, alignment { .start_aligned = true }>);
register_bench<Op, arch>("simd-map/unaligned", bench_map_unary<Op, unaligned_alloc, arch>);
if constexpr (std::is_same_v<typename Op::input_t, typename Op::output_t>)
{
register_bench<Op, arch>("simd-transform/aligned", bench_transform<Op, aligned_alloc, arch>);
register_bench<Op, arch>("simd-transform/unaligned", bench_transform<Op, unaligned_alloc, arch>);
}
register_bench<Op, arch>(
"hot/scalar", bench_scalar<Op, aligned_alloc>, /* sizes = */ { 1024 });
register_bench<Op, arch>(
"hot/simd",
bench_map_unary<Op, aligned_alloc, arch, alignment { .start_aligned = true, .end_aligned = true }>,
/* sizes = */ { 1024 });
}

bool const registered = []
Expand All @@ -108,6 +54,8 @@ namespace
register_benches<xsimd::test::abs_op<double>>();
register_benches<xsimd::test::exp_op<float>>();
register_benches<xsimd::test::exp_op<double>>();
register_benches<xsimd::test::widen_op<std::int8_t>>();
register_benches<xsimd::test::widen_op<std::int16_t>>();
register_benches<xsimd::test::widen_op<std::int32_t>>();
return true;
}();
Expand Down
86 changes: 86 additions & 0 deletions benchmark/map_unary_utils.hpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
/****************************************************************************
* Copyright (c) xsimd-algorithm contributors *
* *
* Distributed under the terms of the BSD 3-Clause License. *
* *
* The full license is in the file LICENSE, distributed with this software. *
****************************************************************************/

#include <cstddef>
#include <cstdint>
#include <format>
#include <string_view>
#include <vector>

#include <benchmark/benchmark.h>

#include "bench_utils.hpp"
#include "xsimd_algorithm/builder.hpp"
#include "xsimd_test_utils/utils.hpp"

namespace xsimd::bench
{
using xsimd::builder::alignment;

template <typename Op, typename Alloc, typename Apply>
void bench_unary(benchmark::State& state, Apply apply)
{
using input_t = typename Op::input_t;

auto const size = static_cast<std::size_t>(state.range(0));
auto [input, output] = Op::template make_input_output<Alloc>(size);

for (auto _ : state)
{
apply(xsimd::test::as_span(input), xsimd::test::as_span(output));
benchmark::DoNotOptimize(output.data());
benchmark::ClobberMemory();
}

state.SetItemsProcessed(static_cast<std::int64_t>(state.iterations() * size));
state.SetBytesProcessed(
static_cast<std::int64_t>(state.iterations() * size * 2 * sizeof(input_t)));
}

template <typename Op, typename Alloc, typename Arch, alignment aligned = alignment {}>
void bench_map_unary(benchmark::State& state)
{
bench_unary<Op, Alloc>(
state,
[](auto in, auto out)
{ Op::template range_apply_map_unary<aligned, Arch>(in, out); });
}

template <typename Op, typename Alloc, typename Arch>
void bench_transform(benchmark::State& state)
{
bench_unary<Op, Alloc>(
state,
[](auto in, auto out)
{ Op::template range_apply_transform<Arch>(in, out); });
}

template <typename Op, typename Alloc>
void bench_scalar(benchmark::State& state)
{
bench_unary<Op, Alloc>(state, [](auto in, auto out)
{ Op::range_apply_scalar(in, out); });
}

template <typename Op, typename Arch, typename Bench>
void register_bench(
std::string_view variant,
Bench bench_fn,
std::vector<std::int64_t> const& sizes = xsimd::bench::bench_sizes<typename Op::input_t>())
{
using input_t = typename Op::input_t;

auto* bench = benchmark::RegisterBenchmark(
std::format("{}/{}/{}/{}", Arch::name(), Op::name, xsimd::bench::type_name<input_t>(), variant),
bench_fn);
for (auto const size : sizes)
{
bench->Arg(size);
}
}
}