51 lines
2 KiB
Rust
51 lines
2 KiB
Rust
//! Bandwidth bench — sequential vs random, MLP
|
|
//! Covers hpc/cpu-cache/bandwidth, mlp, prefetching
|
|
use criterion::{criterion_group, criterion_main, Criterion, BenchmarkId, Throughput};
|
|
|
|
fn bench_sequential(c: &mut Criterion) {
|
|
let mut group = c.benchmark_group("bandwidth_sequential");
|
|
for &n in &[1_000_000, 8_000_000, 32_000_000] {
|
|
let data = vec![0u64; n];
|
|
group.throughput(Throughput::Bytes((n * 8) as u64));
|
|
group.bench_with_input(BenchmarkId::from_parameter(n), &n, |b, _| {
|
|
b.iter(|| {
|
|
let mut sum = 0u64;
|
|
// Auto-vectorizable sequential scan
|
|
for &v in &data { sum = sum.wrapping_add(v); }
|
|
std::hint::black_box(sum)
|
|
});
|
|
});
|
|
}
|
|
group.finish();
|
|
}
|
|
|
|
fn bench_random_mlp(c: &mut Criterion) {
|
|
// MLP: independent loads can overlap. Measure with `prefetch` vs without.
|
|
let mut group = c.benchmark_group("mlp_independent_loads");
|
|
for &n in &[1_000_000, 4_000_000] {
|
|
let data = vec![1u64; n];
|
|
let idxs: Vec<usize> = (0..n).step_by(64).collect(); // cache-line stride
|
|
group.bench_with_input(BenchmarkId::new("without_prefetch", n), &n, |b, _| {
|
|
b.iter(|| {
|
|
let mut sum = 0u64;
|
|
for &i in &idxs { sum = sum.wrapping_add(data[i]); }
|
|
std::hint::black_box(sum)
|
|
});
|
|
});
|
|
group.bench_with_input(BenchmarkId::new("with_prefetch", n), &n, |b, _| {
|
|
b.iter(|| {
|
|
let mut sum = 0u64;
|
|
for &i in &idxs {
|
|
#[cfg(target_arch = "x86_64")]
|
|
unsafe { core::arch::x86_64::_mm_prefetch(data.as_ptr().add(i+128) as *const i8, core::arch::x86_64::_MM_HINT_T0); }
|
|
sum = sum.wrapping_add(data[i]);
|
|
}
|
|
std::hint::black_box(sum)
|
|
});
|
|
});
|
|
}
|
|
group.finish();
|
|
}
|
|
|
|
criterion_group!(benches, bench_sequential, bench_random_mlp);
|
|
criterion_main!(benches);
|