From 2eafe1324640d191ab1b3b102db857ae0384cb14 Mon Sep 17 00:00:00 2001 From: "tembo[bot]" <208362400+tembo-io[bot]@users.noreply.github.com> Date: Tue, 28 Jul 2026 11:52:13 +0000 Subject: [PATCH 1/2] test(bench): add criterion host/rayon baseline --- .github/workflows/benchmarks.yml | 25 +++ Cargo.toml | 7 + README.md | 5 +- benches/host_benchmarks.rs | 269 +++++++++++++++++++++++++++ docs/benchmarking/README.md | 179 ++++++++++++++++++ docs/benchmarking/initial-results.md | 127 +++++++++++++ tests/bench_determinism.rs | 51 +++++ tests/benchmark_support/mod.rs | 181 ++++++++++++++++++ 8 files changed, 843 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/benchmarks.yml create mode 100644 benches/host_benchmarks.rs create mode 100644 docs/benchmarking/README.md create mode 100644 docs/benchmarking/initial-results.md create mode 100644 tests/bench_determinism.rs create mode 100644 tests/benchmark_support/mod.rs diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml new file mode 100644 index 0000000..13ab12c --- /dev/null +++ b/.github/workflows/benchmarks.yml @@ -0,0 +1,25 @@ +name: benchmarks + +on: + push: + branches: [main] + pull_request: + +jobs: + bench-compile: + name: Compile host benchmarks (no-run) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: dtolnay/rust-toolchain@stable + with: + components: rustfmt + - name: Format check + run: cargo fmt --all -- --check + - name: Host-only tests + run: cargo test --no-default-features + # Compile the benchmark suite without running it. Timing-sensitive + # benchmarks are intentionally not executed on shared CI runners; see + # docs/benchmarking/README.md for how to run the suite locally. + - name: Compile benchmarks (no-run, no-default-features) + run: cargo bench --no-run --no-default-features diff --git a/Cargo.toml b/Cargo.toml index de8bfb0..77180b6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -34,3 +34,10 @@ rand = "0.8" rayon = "1.10" smallvec = "1.13" transpose = "0.2" + +[dev-dependencies] +criterion = { version = "0.5", default-features = false, features = ["cargo_bench_support"] } + +[[bench]] +name = "host_benchmarks" +harness = false diff --git a/README.md b/README.md index 784e2c1..3ee7acc 100644 --- a/README.md +++ b/README.md @@ -12,4 +12,7 @@ OpenCL is a trademark of Apple Inc. used by permission by the Khronos Group. For - This excellent overview of OpenCL kernel programming & optimization: https://www.nersc.gov/assets/pubs_presos/MattsonTutorialSC14.pdf - - A benchmarking tool available for comparing numpy, ndarray and ha-ndarray is available in the `benchmark` branch and can be built with `cargo run --bin benchmark --features benchmark` see [README.md](./benchmark/README.md) for more information. + - A reproducible Criterion benchmark baseline for the host/Rayon platform lives in the `benches/` + directory and can be run with `cargo bench --no-default-features`. See + [docs/benchmarking/README.md](./docs/benchmarking/README.md) for the benchmark inventory, local + execution instructions, and how future CubeCL/OpenCL groups should reuse this baseline. diff --git a/benches/host_benchmarks.rs b/benches/host_benchmarks.rs new file mode 100644 index 0000000..254ece2 --- /dev/null +++ b/benches/host_benchmarks.rs @@ -0,0 +1,269 @@ +//! Host/Rayon Criterion benchmark baseline for `ha-ndarray`. +//! +//! This benchmark target is deliberately built against the **current public +//! API** of `ha-ndarray` with `--no-default-features` (host-only, Rayon). It +//! establishes a reproducible baseline on `main` before the CubeCL HAL +//! migration changes execution behavior. +//! +//! # Operation inventory +//! +//! | Category | ha-ndarray public API | Benchmark group | +//! |---------------------|------------------------------------------------------|--------------------------| +//! | 1. construction | `ArrayBuf::convert(&data, shape)` (alloc + copy) | `construct_convert` | +//! | 2. unary elementwise| `NDArrayUnary::exp` | `unary_exp` | +//! | 3. binary elementwise| `NDArrayMath::add` (same-shape) | `binary_add` | +//! | 3. binary broadcast | `NDArrayTransform::broadcast` + `NDArrayMath::add` | `binary_add_broadcast` | +//! | 4. reduction | `NDArrayReduceAll::sum_all` | `reduce_sum_all` | +//! | 5. matrix multiply | `MatrixDual::matmul` | `matmul` | +//! | 6. non-contig. view | `NDArrayTransform::transpose` (materialized view) | `transpose` | +//! +//! `exp` is chosen as the unary op because it exercises a transcendental +//! elementwise kernel on `f32`. `add` is chosen as the binary op, with the +//! broadcast variant broadcasting a `[1, N]` operand up to `[M, N]` before the +//! add. `sum_all` is chosen as the reduction because it is available on +//! borrowed (non-`'static`) inputs via the public `NDArrayReduceAll` trait; +//! axis-reduction (`NDArrayReduce::sum`) requires an owned `'static` accessor +//! and is documented as a future addition. `transpose` materializes a +//! non-contiguous view (swapped strides) into a contiguous buffer. +//! +//! # Shape matrix +//! +//! Elementwise / reduction / transform benchmarks use a `dtype x shape` matrix: +//! * small (latency): `[16, 16]` = 256 elements +//! * medium: `[256, 256]` = 65 536 elements +//! * large (throughput):`[1024, 1024]`= 1 048 576 elements +//! +//! `matmul` uses square matrices sized for a bounded budget: +//! * small: `[16, 16] x [16, 16]` (~4 K MACs) +//! * medium:`[96, 96] x [96, 96]` (~885 K MACs) +//! * large: `[256, 256] x [256, 256]`(~16.8 M MACs) +//! +//! # Measurement model +//! +//! `ha-ndarray` operations are lazy: building an op tree (e.g. `input.exp()`) +//! is essentially free, and computation only happens when the result is +//! materialized via `NDArrayRead::buffer` (or `into_read`). Operation-only +//! benchmarks therefore construct deterministic input data **once, outside the +//! measured region**, build a cheap borrowed view of that data per iteration in +//! the (untimed) `iter_batched` setup, and measure op construction + +//! materialization. The `construct_convert` group is explicitly end-to-end and +//! includes allocation + copy. +//! +//! Inputs are reused across iterations (kept in cache), which is standard for +//! in-process micro-benchmarks and does not change operation semantics. + +#[path = "../tests/benchmark_support/mod.rs"] +mod support; + +use criterion::{criterion_group, BatchSize, Criterion}; +use ha_ndarray::*; + +/// Elementwise / reduction / transform shape matrix: `(rows, cols)`. +const ELEM_SIZES: &[(usize, usize)] = &[(16, 16), (256, 256), (1024, 1024)]; + +/// `matmul` shape matrix: square `(k, k)` operands. +const MATMUL_SIZES: &[(usize, usize)] = &[(16, 16), (96, 96), (256, 256)]; + +/// Build a borrowed, buffer-backed input view over `data` with the given +/// `shape`. Construction is cheap (a slice reference + a cloned `SmallVec` +/// shape) and is used in the untimed `iter_batched` setup region. +fn input_view(data: &[f32], shape: Shape) -> ArrayBuf { + ArrayBuf::new(data, shape).expect("construct input view") +} + +/// Category 1: array construction / allocation (end-to-end). +/// +/// Measures `ArrayBuf::convert(&data, shape)`, which allocates a host buffer +/// and copies the deterministic source slice into it. This is the only group +/// that intentionally includes allocation cost. +fn bench_construct(c: &mut Criterion) { + let mut group = c.benchmark_group("construct_convert"); + for &(m, n) in ELEM_SIZES { + let data = support::uniform_f32(m * n, support::FIXTURE_SEED); + let shape = shape![m, n]; + let id = format!("f32/{}", support::shape_id(&[m, n])); + group.bench_function(id, |b| { + b.iter(|| { + let arr: ArrayBuf> = + ArrayBuf::convert(&data[..], shape.clone()).expect("convert"); + criterion::black_box(arr); + }); + }); + } + group.finish(); +} + +/// Category 2: unary elementwise operation (`exp`), operation-only. +fn bench_unary(c: &mut Criterion) { + let mut group = c.benchmark_group("unary_exp"); + for &(m, n) in ELEM_SIZES { + let data = support::uniform_f32(m * n, support::FIXTURE_SEED); + let shape = shape![m, n]; + let id = format!("f32/{}", support::shape_id(&[m, n])); + group.bench_function(id, |b| { + b.iter_batched( + || input_view(&data, shape.clone()), + |input| { + let out = input.exp().expect("exp"); + criterion::black_box(out.buffer().expect("materialize")); + }, + BatchSize::SmallInput, + ); + }); + } + group.finish(); +} + +/// Category 3a: binary elementwise operation (`add`), same-shape, operation-only. +fn bench_binary(c: &mut Criterion) { + let mut group = c.benchmark_group("binary_add"); + for &(m, n) in ELEM_SIZES { + let left = support::uniform_f32(m * n, support::FIXTURE_SEED); + let right = support::uniform_f32(m * n, support::FIXTURE_SEED.wrapping_add(1)); + let shape = shape![m, n]; + let id = format!("f32/{}", support::shape_id(&[m, n])); + group.bench_function(id, |b| { + b.iter_batched( + || { + ( + input_view(&left, shape.clone()), + input_view(&right, shape.clone()), + ) + }, + |(l, r)| { + let out = l.add(r).expect("add"); + criterion::black_box(out.buffer().expect("materialize")); + }, + BatchSize::SmallInput, + ); + }); + } + group.finish(); +} + +/// Category 3b: binary elementwise operation (`add`) with a broadcast operand. +/// +/// Broadcasts a `[1, N]` operand up to `[M, N]` then adds, exercising the +/// `Transform::broadcast` view within the measured region. +fn bench_binary_broadcast(c: &mut Criterion) { + let mut group = c.benchmark_group("binary_add_broadcast"); + for &(m, n) in ELEM_SIZES { + let left = support::uniform_f32(m * n, support::FIXTURE_SEED); + let right = support::uniform_f32(n, support::FIXTURE_SEED.wrapping_add(2)); + let l_shape = shape![m, n]; + let r_shape = shape![1, n]; + let b_shape = shape![m, n]; + let id = format!("f32/{}", support::shape_id(&[m, n])); + group.bench_function(id, |b| { + b.iter_batched( + || { + ( + input_view(&left, l_shape.clone()), + input_view(&right, r_shape.clone()), + ) + }, + |(l, r)| { + let r_b = r.broadcast(b_shape.clone()).expect("broadcast"); + let out = l.add(r_b).expect("add"); + criterion::black_box(out.buffer().expect("materialize")); + }, + BatchSize::SmallInput, + ); + }); + } + group.finish(); +} + +/// Category 4: reduction (`sum_all`), operation-only. +/// +/// Reduces the entire array to a single `f32` scalar via the public +/// `NDArrayReduceAll::sum_all` trait method, which works on borrowed inputs. +fn bench_reduce(c: &mut Criterion) { + let mut group = c.benchmark_group("reduce_sum_all"); + for &(m, n) in ELEM_SIZES { + let data = support::uniform_f32(m * n, support::FIXTURE_SEED); + let shape = shape![m, n]; + let id = format!("f32/{}", support::shape_id(&[m, n])); + group.bench_function(id, |b| { + b.iter_batched( + || input_view(&data, shape.clone()), + |input| { + let sum = input.sum_all().expect("sum_all"); + criterion::black_box(sum); + }, + BatchSize::SmallInput, + ); + }); + } + group.finish(); +} + +/// Category 5: matrix multiplication, operation-only. +fn bench_matmul(c: &mut Criterion) { + let mut group = c.benchmark_group("matmul"); + for &(k, _) in MATMUL_SIZES { + let left = support::uniform_f32(k * k, support::FIXTURE_SEED); + let right = support::uniform_f32(k * k, support::FIXTURE_SEED.wrapping_add(3)); + let l_shape = shape![k, k]; + let r_shape = shape![k, k]; + let id = format!("f32/{}", support::shape_id(&[k, k])); + group.bench_function(id, |b| { + b.iter_batched( + || { + ( + input_view(&left, l_shape.clone()), + input_view(&right, r_shape.clone()), + ) + }, + |(l, r)| { + let out = l.matmul(r).expect("matmul"); + criterion::black_box(out.buffer().expect("materialize")); + }, + BatchSize::SmallInput, + ); + }); + } + group.finish(); +} + +/// Category 6: non-contiguous view / transform (`transpose`), operation-only. +/// +/// Transposes a 2-D array (swapping strides to produce a non-contiguous view) +/// and materializes it into a contiguous buffer. +fn bench_transpose(c: &mut Criterion) { + let mut group = c.benchmark_group("transpose"); + for &(m, n) in ELEM_SIZES { + let data = support::uniform_f32(m * n, support::FIXTURE_SEED); + let shape = shape![m, n]; + let id = format!("f32/{}", support::shape_id(&[m, n])); + group.bench_function(id, |b| { + b.iter_batched( + || input_view(&data, shape.clone()), + |input| { + let out = input.transpose(None).expect("transpose"); + criterion::black_box(out.buffer().expect("materialize")); + }, + BatchSize::SmallInput, + ); + }); + } + group.finish(); +} + +criterion_group! { + name = benches; + config = Criterion::default() + .sample_size(support::SAMPLE_SIZE) + .warm_up_time(support::WARMUP_TIME) + .measurement_time(support::MEASUREMENT_TIME); + targets = bench_construct, bench_unary, bench_binary, bench_binary_broadcast, bench_reduce, bench_matmul, bench_transpose, +} + +fn main() { + let env = support::Environment::probe("no-default-features (host-only, Rayon)"); + eprintln!("[ha-ndarray benchmark environment]\n{}", env.report()); + benches(); + criterion::Criterion::default() + .configure_from_args() + .final_summary(); +} diff --git a/docs/benchmarking/README.md b/docs/benchmarking/README.md new file mode 100644 index 0000000..62cf378 --- /dev/null +++ b/docs/benchmarking/README.md @@ -0,0 +1,179 @@ +# Benchmarking ha-ndarray + +This directory documents the reproducible Criterion benchmark baseline for +`ha-ndarray`, established on `main` before the CubeCL HAL migration changes +execution behavior (issue #22). + +The baseline is **host-only / Rayon** and is built against the current public +API with `--no-default-features`. It is intentionally not tied to OpenCL, +GPU, or any accelerator. + +## Layout + +| Path | Purpose | +|---------------------------------------|--------------------------------------------------------------| +| `benches/host_benchmarks.rs` | Criterion targets for the six benchmark categories. | +| `tests/benchmark_support/mod.rs` | Shared deterministic fixture + environment-recording utility.| +| `tests/bench_determinism.rs` | Tests that the fixture is deterministic and in range. | +| `docs/benchmarking/initial-results.md`| The initial result artifact + exact environment identity. | + +## Benchmark inventory and operation mapping + +`ha-ndarray` operations are **lazy**: building an op tree (e.g. `input.exp()`) +is essentially free, and computation only happens when a result is materialized +via `NDArrayRead::buffer` (or `into_read`). Operation-only benchmarks therefore +build deterministic input data once, outside the measured region, and measure +op construction + materialization. `criterion::black_box` is used to prevent the +compiler from eliminating work. + +| # | Category | Public API used | Benchmark group | End-to-end? | +|---|-----------------------|--------------------------------------------------|--------------------------|-------------| +| 1 | construction | `ArrayBuf::convert(&data, shape)` (alloc + copy) | `construct_convert` | yes | +| 2 | unary elementwise | `NDArrayUnary::exp` | `unary_exp` | no | +| 3 | binary elementwise | `NDArrayMath::add` (same-shape) | `binary_add` | no | +| 3 | binary (broadcast) | `NDArrayTransform::broadcast` + `NDArrayMath::add`| `binary_add_broadcast` | no | +| 4 | reduction | `NDArrayReduceAll::sum_all` | `reduce_sum_all` | no | +| 5 | matrix multiplication | `MatrixDual::matmul` | `matmul` | no | +| 6 | non-contiguous view | `NDArrayTransform::transpose` (materialized) | `transpose` | no | + +### Chosen operations and substitutions + +- `exp` is the unary op: it exercises a transcendental elementwise `f32` kernel. +- `add` is the binary op; the broadcast variant broadcasts a `[1, N]` operand up + to `[M, N]` before the add, exercising `Transform::broadcast` in the measured + region. +- `sum_all` is the reduction. It is available on **borrowed** (non-`'static`) + inputs via the public `NDArrayReduceAll` trait. Axis-reduction + (`NDArrayReduce::sum`) requires an owned `'static` accessor (the + `Accessor::from(A)` indirection used by `reduce_axes` has a `B: 'static` + bound), so it is documented as a future addition rather than substituted with + extra allocation in this baseline. +- `transpose` is the non-contiguous-view transform: it swaps strides to produce + a non-contiguous view, then materializes it into a contiguous buffer. + +No product functionality was added to make a benchmark possible. + +## Shape matrix + +Benchmark identifiers encode `dtype/shape` (e.g. `matmul/f32/256x256`) so that +results are machine-readable and comparable across runs on the same machine. + +Elementwise / reduction / transform benchmarks: + +| Case | Shape | Elements | Orientation | +|--------|----------------|-------------|-----------------| +| small | `[16, 16]` | 256 | latency | +| medium | `[256, 256]` | 65 536 | representative | +| large | `[1024, 1024]` | 1 048 576 | throughput | + +`matmul` uses square operands sized for a bounded budget: + +| Case | Operands | ~MACs | +|--------|--------------------------------|-----------| +| small | `[16, 16] x [16, 16]` | 4 K | +| medium | `[96, 96] x [96, 96]` | 885 K | +| large | `[256, 256] x [256, 256]` | 16.8 M | + +The large cases are bounded to remain safe in the documented Tembo environment +(see `initial-results.md` for the measured total duration). + +## Measurement rules + +- **Deterministic inputs**: all inputs are generated from a fixed seed + (`FIXTURE_SEED`) by `tests/benchmark_support/mod.rs`, producing byte-identical + data on every run and machine. Determinism is verified by + `tests/bench_determinism.rs`. +- **Setup is not measured**: input data is constructed once outside the + benchmark; per-iteration input views are built in the untimed `iter_batched` + setup region. The only intentionally end-to-end group is `construct_convert`. +- **`black_box`**: every benchmark feeds its result through + `criterion::black_box` to prevent optimization from eliminating work. +- **Bounded budget**: the Criterion config uses `sample_size = 10`, + `warm_up_time = 1 s`, `measurement_time = 3 s` (see + `tests/benchmark_support/mod.rs`). With 21 benchmarks this gives a documented, + finite total run time recorded in `initial-results.md`. +- **Inputs are reused** across iterations (kept in cache). This is standard for + in-process micro-benchmarks and does not change operation semantics. + +## How to run locally + +Compile only (the CI path; safe on shared runners): + +``` +cargo bench --no-run --no-default-features +``` + +Run the full host-only suite: + +``` +cargo bench --no-default-features +``` + +Run a single group or benchmark (Criterion filter): + +``` +cargo bench --no-default-features -- matmul +cargo bench --no-default-features -- "matmul/f32/256x256" +``` + +Criterion writes machine-readable output under `target/criterion/` (each +benchmark has `new/estimates.json` with mean/median/slope/throughput, plus +`benchmark.json`). The benchmark binary also prints the recorded environment to +stderr at start-up. + +## Recording the environment + +The environment is recorded automatically at the start of every benchmark run +via `support::Environment::probe` and printed to stderr. It captures: + +- `rustc` / `cargo` versions +- OS and architecture +- CPU model (parsed from `/proc/cpuinfo` on Linux) +- logical and physical core counts (`num_cpus`) +- the active feature set + +When publishing a result, copy this block verbatim into the result record (as +`initial-results.md` does) so the result is tied to an exact machine identity. + +## Comparing results + +- **Do not compare numbers from different machines as though they are directly + equivalent.** CPU model, core count, frequency, and memory bandwidth differ. +- Compare a new run against a previous run **on the same machine** using + Criterion's own comparison (`cargo bench` reuses `target/criterion/` baselines + and reports regressions/improvements), or by diffing `estimates.json`. +- A faster or slower number is not, by itself, evidence of correctness. + +## How future CubeCL/OpenCL groups should reuse this baseline (issues #34-#50) + +The host baseline is intentionally stable and must not change when accelerator +groups are added: + +1. **Do not modify the host benchmark groups** (`construct_convert`, + `unary_exp`, `binary_add`, `binary_add_broadcast`, `reduce_sum_all`, + `matmul`, `transpose`) or the shared fixture. Add new groups alongside them. +2. **Reuse the shared fixture** (`tests/benchmark_support/mod.rs`) so that + CubeCL/OpenCL groups consume byte-identical inputs to the host groups. This is + what makes an accelerator-vs-host comparison meaningful on the same machine. +3. **Mirror the shape matrix and naming** so that a future `matmul/cubecl/...` + group can be compared against `matmul/f32/...` on the same machine. +4. **Gate accelerator groups behind their feature** (for example + `#[cfg(feature = "opencl")]`), and document any all-feature/OpenCL validation + that was not possible in this baseline. +5. **Record the same environment block** so accelerator results carry exact + machine identity. + +This baseline does not introduce any performance threshold, supported-hardware +claim, or backend-retirement decision. + +## Known limitations + +- No OpenCL, GPU, or `all`-feature benchmarking is performed or validated here; + that is an explicit non-goal of this baseline and is left to the CubeCL/OpenCL + groups in #34-#50. +- Axis-reduction (`NDArrayReduce::sum`) is not benchmarked because it requires + an owned `'static` accessor through the public `reduce_axes` path; `sum_all` + is used as the representative reduction instead. +- `matmul` does not exercise batched or broadcasted matrix multiply at the large + size; only the square case is in the baseline shape matrix. +- Results are machine-specific and must not be compared across machines. diff --git a/docs/benchmarking/initial-results.md b/docs/benchmarking/initial-results.md new file mode 100644 index 0000000..fcb110e --- /dev/null +++ b/docs/benchmarking/initial-results.md @@ -0,0 +1,127 @@ +# Initial benchmark result artifact (issue #22) + +This is the initial Criterion result artifact for the host/Rayon baseline. It +records the exact environment, the exact commands executed, the benchmark +results, the total duration, and the test results. + +> **These numbers are specific to the machine below and must not be compared +> directly with numbers from any other machine.** They are a baseline, not a +> performance threshold or a supported-hardware claim. + +## Exact environment identity + +Recorded automatically by `support::Environment::probe` and printed to stderr at +the start of the run: + +``` +rustc: rustc 1.96.0 (ac68faa20 2026-05-25) +cargo: cargo 1.96.0 (30a34c682 2026-05-25) +os: linux +arch: x86_64 +cpu_model: Intel(R) Xeon(R) Platinum 8275CL CPU @ 3.00GHz +logical_cores: 2 +physical_cores: 2 +feature_set: no-default-features (host-only, Rayon) +``` + +## Exact commands executed + +```text +cargo fmt --all -- --check +cargo test --no-default-features +cargo bench --no-run --no-default-features +cargo bench --no-default-features +``` + +## Benchmark compilation result + +```text +cargo bench --no-run --no-default-features + Finished `bench` profile [optimized] target(s) + Executable benches/host_benchmarks.rs (target/release/deps/host_benchmarks-...) +``` + +Compilation succeeds in a clean checkout with `--no-default-features`. + +## Results + +Criterion reports each benchmark as `[lower_bound mean upper_bound]`. The +machine-readable source of truth for every benchmark is +`target/criterion///new/estimates.json` (not committed to the +repository); the values below are taken from that run. + +| Benchmark | Time [lower mean upper] | +|------------------------------------|----------------------------------| +| construct_convert/f32/16x16 | [88.469 ns 89.498 ns 90.417 ns] | +| construct_convert/f32/256x256 | [7.3174 µs 7.4304 µs 7.5423 µs] | +| construct_convert/f32/1024x1024 | [344.94 µs 345.59 µs 346.43 µs] | +| unary_exp/f32/16x16 | [10.921 µs 12.143 µs 13.931 µs] | +| unary_exp/f32/256x256 | [162.21 µs 165.19 µs 168.11 µs] | +| unary_exp/f32/1024x1024 | [2.2983 ms 2.3161 ms 2.3318 ms] | +| binary_add/f32/16x16 | [26.883 µs 27.760 µs 28.377 µs] | +| binary_add/f32/256x256 | [78.554 µs 79.549 µs 80.895 µs] | +| binary_add/f32/1024x1024 | [849.29 µs 857.92 µs 872.34 µs] | +| binary_add_broadcast/f32/16x16 | [41.802 µs 44.219 µs 45.955 µs] | +| binary_add_broadcast/f32/256x256 | [651.33 µs 661.02 µs 671.35 µs] | +| binary_add_broadcast/f32/1024x1024 | [9.7063 ms 9.8600 ms 10.049 ms] | +| reduce_sum_all/f32/16x16 | [11.767 µs 12.015 µs 12.203 µs] | +| reduce_sum_all/f32/256x256 | [52.757 µs 53.330 µs 53.970 µs] | +| reduce_sum_all/f32/1024x1024 | [616.56 µs 623.98 µs 631.21 µs] | +| matmul/f32/16x16 | [47.459 µs 48.198 µs 49.180 µs] | +| matmul/f32/96x96 | [1.5172 ms 1.5369 ms 1.5575 ms] | +| matmul/f32/256x256 | [14.218 ms 14.574 ms 14.998 ms] | +| transpose/f32/16x16 | [24.586 µs 24.774 µs 24.931 µs] | +| transpose/f32/256x256 | [562.52 µs 568.67 µs 574.14 µs] | +| transpose/f32/1024x1024 | [12.287 ms 12.435 ms 12.686 ms] | + +Observations (not thresholds): + +- The small (`16x16`) cases are latency-dominated by Rayon thread-pool dispatch + (tens of µs), which is the intended latency-oriented characterization. +- The large elementwise/transform cases are throughput/memory-bound; `matmul` + and `transpose` at `1024x1024`/`256x256` are the most compute-intensive. +- `binary_add_broadcast` is slower than same-shape `binary_add` because the + broadcast operand is materialized (gathered with broadcast strides) within the + measured region before the add. + +## Total benchmark duration + +```text +cargo bench --no-default-features # 21 benchmarks, sample_size=10, + # warm_up_time=1s, measurement_time=3s +Total wall time: 103 s +``` + +## Test results + +```text +cargo test --no-default-features +``` + +All host-only tests pass, including the 5 new `bench_determinism` tests that +verify the fixture is deterministic, in range, and seed-distinct: + +- `arithmetic` 2 passed +- `bench_determinism` 5 passed +- `compare` 7 passed +- `cond` 1 passed +- `construct` 3 passed +- `linalg` 2 passed +- `reduce` 3 passed +- `transform` 9 passed + +## Known limitations and unavailable validation + +- **No OpenCL / `all`-feature validation.** This baseline is host-only + (`--no-default-features`). OpenCL/GPU/`all`-feature benchmarking is an + explicit non-goal and is left to the CubeCL/OpenCL groups in #34-#50. +- **Axis-reduction is not benchmarked.** `NDArrayReduce::sum` (reduce over an + axis) requires an owned `'static` accessor via the public `reduce_axes` path; + `NDArrayReduceAll::sum_all` is used as the representative reduction instead. +- **Batched/broadcasted `matmul`** is not in the baseline shape matrix; only the + square case is measured. +- **Results are machine-specific.** The numbers above are tied to the + environment block at the top of this file and must not be compared with + results from a different machine. +- No performance threshold, supported-hardware claim, or backend-retirement + decision is introduced by this artifact. diff --git a/tests/bench_determinism.rs b/tests/bench_determinism.rs new file mode 100644 index 0000000..5a84ef9 --- /dev/null +++ b/tests/bench_determinism.rs @@ -0,0 +1,51 @@ +//! Verifies that the shared benchmark fixture is deterministic and within the +//! documented numeric range. This guards the "deterministic inputs" acceptance +//! criterion of issue #22 independently of running the timing suite. + +#[path = "benchmark_support/mod.rs"] +mod support; + +#[test] +fn test_fixture_is_deterministic() { + for &size in &[256, 65_536, 1_048_576] { + let first = support::uniform_f32(size, support::FIXTURE_SEED); + let second = support::uniform_f32(size, support::FIXTURE_SEED); + assert_eq!(first, second, "fixture is not reproducible for size {size}"); + } +} + +#[test] +fn test_fixture_distinct_seeds_differ() { + let a = support::uniform_f32(1024, support::FIXTURE_SEED); + let b = support::uniform_f32(1024, support::FIXTURE_SEED.wrapping_add(1)); + assert_ne!(a, b, "distinct seeds produced identical data"); +} + +#[test] +fn test_fixture_respects_range() { + let data = support::uniform_f32(4096, support::FIXTURE_SEED); + for v in data { + assert!( + v >= support::FIXTURE_LO && v <= support::FIXTURE_HI, + "fixture value {v} outside documented range" + ); + } +} + +#[test] +fn test_fixture_nonzero_variant() { + let eps = 0.25; + let data = support::uniform_f32_nonzero(4096, support::FIXTURE_SEED, eps); + assert!( + data.iter().copied().all(|v| v.abs() >= eps), + "nonzero fixture produced a value within the exclusion band" + ); +} + +#[test] +fn test_environment_probe_succeeds() { + let env = support::Environment::probe("test"); + assert!(!env.os.is_empty()); + assert!(env.logical_cores > 0); + assert_eq!(env.feature_set, "test"); +} diff --git a/tests/benchmark_support/mod.rs b/tests/benchmark_support/mod.rs new file mode 100644 index 0000000..c12fc4d --- /dev/null +++ b/tests/benchmark_support/mod.rs @@ -0,0 +1,181 @@ +//! Shared deterministic fixture and environment-recording utilities for the +//! ha-ndarray Criterion benchmark suite. +//! +//! This module is deliberately decoupled from the `ha-ndarray` public API: it +//! only produces raw `Vec` data and records host environment metadata. The +//! benchmark targets and the determinism test construct `ha-ndarray` arrays from +//! this data so that input construction is excluded from operation-only +//! measurements. +//! +//! It is intended to be included verbatim from two compilation contexts via +//! `#[path]`: +//! +//! * `benches/host_benchmarks.rs` - `#[path = "../tests/benchmark_support/mod.rs"] mod support;` +//! * `tests/bench_determinism.rs` - `#[path = "benchmark_support/mod.rs"] mod support;` +//! +//! Both contexts have access to the `rand` and `num_cpus` crate dependencies of +//! `ha-ndarray`, so this module depends on nothing else. + +use std::time::Duration; + +use rand::rngs::StdRng; +use rand::{Rng, SeedableRng}; + +/// The fixed seed used for all benchmark input generation. +/// +/// Every benchmark input is derived from this seed so that a given shape always +/// produces byte-identical data on every run and every machine. +pub const FIXTURE_SEED: u64 = 0x6e64_6172_6179_0001; + +/// The inclusive range from which deterministic fixture values are drawn. +/// +/// The bounds are chosen so that every benchmarked operation stays in a safe +/// numeric range for `f32` (for example `exp(5.0) ~= 148.4` is well within the +/// `f32` range, and there are no zeros to cause division-by-zero surprises in +/// downstream comparisons). +pub const FIXTURE_LO: f32 = -5.0; +pub const FIXTURE_HI: f32 = 5.0; + +/// Default Criterion measurement budget used by the benchmark targets. +/// +/// These values are intentionally small so that a complete local run of the +/// host-only suite has a documented, finite wall-clock budget. See +/// `docs/benchmarking/README.md` for the expected total duration. +pub const SAMPLE_SIZE: usize = 10; +pub const WARMUP_TIME: Duration = Duration::from_secs(1); +pub const MEASUREMENT_TIME: Duration = Duration::from_secs(3); + +/// Generate a deterministic `Vec` of `size` elements, drawn uniformly from +/// `[FIXTURE_LO, FIXTURE_HI]`. +/// +/// The same `(size, seed)` pair always returns the same data, independent of +/// host, thread count, or run order. A different `seed` produces an independent +/// stream, which is useful for generating distinct operands (for example the two +/// sides of a binary operation). +pub fn uniform_f32(size: usize, seed: u64) -> Vec { + let mut rng = StdRng::seed_from_u64(seed); + (0..size) + .map(|_| rng.gen_range(FIXTURE_LO..=FIXTURE_HI)) + .collect() +} + +/// Generate a deterministic `Vec` of `size` elements with no zero values, +/// drawn from `[FIXTURE_LO, FIXTURE_HI]` but with any value whose absolute value +/// is below `eps` shifted away from zero. +/// +/// Useful for operands that must avoid exact zeros (for example the denominator +/// of a division, or the base of a logarithm) while remaining deterministic. +/// Not consumed by the current host benchmarks, but provided for future groups +/// (for example a CubeCL division or logarithm benchmark) and validated by the +/// `bench_determinism` integration test. +#[allow(dead_code)] +pub fn uniform_f32_nonzero(size: usize, seed: u64, eps: f32) -> Vec { + let mut rng = StdRng::seed_from_u64(seed); + (0..size) + .map(|_| { + let mut v = rng.gen_range(FIXTURE_LO..=FIXTURE_HI); + if v.abs() < eps { + v = if v >= 0.0 { eps } else { -eps }; + } + v + }) + .collect() +} + +/// Render a shape slice as a compact, machine-readable `d1xd2x...` string used +/// to construct stable benchmark identifiers. +pub fn shape_id(shape: &[usize]) -> String { + shape + .iter() + .map(|d| d.to_string()) + .collect::>() + .join("x") +} + +/// Recorded identity of the host that produced a benchmark result. +/// +/// All fields are strings so that the record can be serialized trivially and +/// compared across runs. None of these fields imply a performance threshold or a +/// supported-hardware claim. +#[derive(Debug, Clone)] +pub struct Environment { + pub rustc: String, + pub cargo: String, + pub os: String, + pub arch: String, + pub cpu_model: String, + pub logical_cores: usize, + pub physical_cores: usize, + pub feature_set: String, +} + +impl Environment { + /// Probe the current host and return its identity. + /// + /// Falling back to the string `"unknown"` for any field that cannot be + /// determined, so that recording never panics on an unsupported platform. + pub fn probe(feature_set: &str) -> Self { + Self { + rustc: command_output(["rustc", "--version"]), + cargo: command_output(["cargo", "--version"]), + os: std::env::consts::OS.to_string(), + arch: std::env::consts::ARCH.to_string(), + cpu_model: cpu_model_name().unwrap_or_else(|| "unknown".to_string()), + logical_cores: num_cpus::get(), + physical_cores: num_cpus::get_physical(), + feature_set: feature_set.to_string(), + } + } + + /// Render the environment as a stable, human-readable block. + pub fn report(&self) -> String { + format!( + "rustc: {}\n\ + cargo: {}\n\ + os: {}\n\ + arch: {}\n\ + cpu_model: {}\n\ + logical_cores: {}\n\ + physical_cores: {}\n\ + feature_set: {}\n", + self.rustc, + self.cargo, + self.os, + self.arch, + self.cpu_model, + self.logical_cores, + self.physical_cores, + self.feature_set, + ) + } +} + +fn command_output(cmd: [&str; 2]) -> String { + std::process::Command::new(cmd[0]) + .arg(cmd[1]) + .output() + .ok() + .and_then(|out| String::from_utf8(out.stdout).ok()) + .map(|s| s.trim().to_string()) + .unwrap_or_else(|| "unknown".to_string()) +} + +/// Read the CPU model name on Linux by parsing `/proc/cpuinfo`. +/// +/// Returns `None` on non-Linux platforms or if the field is absent. +fn cpu_model_name() -> Option { + if std::env::consts::OS != "linux" { + return None; + } + + let contents = std::fs::read_to_string("/proc/cpuinfo").ok()?; + for line in contents.lines() { + if let Some(rest) = line.strip_prefix("model name") { + if let Some((_, value)) = rest.split_once(':') { + return Some(value.trim().to_string()); + } + } + } + + None +} From c80103fe8193009baa4a8fc829d1318a8e3aef7a Mon Sep 17 00:00:00 2001 From: "tembo[bot]" <208362400+tembo-io[bot]@users.noreply.github.com> Date: Tue, 28 Jul 2026 11:57:33 +0000 Subject: [PATCH 2/2] chore(bench): remove benchmarks CI workflow --- .github/workflows/benchmarks.yml | 25 ------------------------- docs/benchmarking/README.md | 2 +- 2 files changed, 1 insertion(+), 26 deletions(-) delete mode 100644 .github/workflows/benchmarks.yml diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml deleted file mode 100644 index 13ab12c..0000000 --- a/.github/workflows/benchmarks.yml +++ /dev/null @@ -1,25 +0,0 @@ -name: benchmarks - -on: - push: - branches: [main] - pull_request: - -jobs: - bench-compile: - name: Compile host benchmarks (no-run) - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - uses: dtolnay/rust-toolchain@stable - with: - components: rustfmt - - name: Format check - run: cargo fmt --all -- --check - - name: Host-only tests - run: cargo test --no-default-features - # Compile the benchmark suite without running it. Timing-sensitive - # benchmarks are intentionally not executed on shared CI runners; see - # docs/benchmarking/README.md for how to run the suite locally. - - name: Compile benchmarks (no-run, no-default-features) - run: cargo bench --no-run --no-default-features diff --git a/docs/benchmarking/README.md b/docs/benchmarking/README.md index 62cf378..3b38703 100644 --- a/docs/benchmarking/README.md +++ b/docs/benchmarking/README.md @@ -97,7 +97,7 @@ The large cases are bounded to remain safe in the documented Tembo environment ## How to run locally -Compile only (the CI path; safe on shared runners): +Compile only (safe to verify the suite builds without running timings): ``` cargo bench --no-run --no-default-features