diff --git a/CHANGELOG.md b/CHANGELOG.md index b7d5061..fd69de7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,18 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +### Added + +- **Arm benchmark data point** (`benchmarks/device-results-neoverse-n1.*`): the + device table on an Ampere Neoverse-N1 host (16 pinned cores) with an RTX + 4060 and an RTX A400. + +### Fixed + +- `benchmarks/bench_devices.py` reported the CPU of Arm hosts as `aarch64`; + it now reads the model from `lscpu` (e.g. `Neoverse-N1`) when + `/proc/cpuinfo` has no model name. + ## [2.4.0] - 2026-10-07 ### Added diff --git a/benchmarks/bench_devices.py b/benchmarks/bench_devices.py index ffb9ba4..b014991 100644 --- a/benchmarks/bench_devices.py +++ b/benchmarks/bench_devices.py @@ -112,6 +112,15 @@ def _cpu_model() -> str: return line.split(":", 1)[1].strip() except OSError: pass + # Arm /proc/cpuinfo has no "model name"; lscpu decodes the part number + # (e.g. "Neoverse-N1"). + try: + output = subprocess.run(["lscpu"], check=True, capture_output=True, text=True).stdout + except (OSError, subprocess.CalledProcessError): + output = "" + for line in output.splitlines(): + if line.startswith("Model name:"): + return line.split(":", 1)[1].strip() return platform.processor() or "unknown CPU" diff --git a/benchmarks/device-results-neoverse-n1-rtx4060.json b/benchmarks/device-results-neoverse-n1-rtx4060.json new file mode 100644 index 0000000..4f93811 --- /dev/null +++ b/benchmarks/device-results-neoverse-n1-rtx4060.json @@ -0,0 +1,221 @@ +{ + "schema_version": 1, + "generated_at_utc": "2026-10-07T07:14:14.461693+00:00", + "command": "/home/jtaylor/miniforge3/envs/aosim-bench/bin/python benchmarks/bench_devices.py --seconds 2 --warmup 10 --device both --output /home/jtaylor/aosim-bench-logs/artifacts/getframes-device-rtx4060.json", + "revision": "7b01daf765d432ab141176548f2b11cf0e7fc11c", + "source_dirty": true, + "python": "3.13.15 | packaged by conda-forge | (main, Sep 2 2026, 22:02:03) [GCC 15.3.0]", + "platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39", + "processor": "aarch64", + "cpu": "Neoverse-N1", + "gpu": "NVIDIA GeForce RTX 4060", + "dependencies": { + "getframes": "2.4.0", + "numpy": "2.5.3", + "scipy": "1.18.1", + "cupy": "14.2.0" + }, + "methodology": { + "seconds_per_cell": 2.0, + "warmup_frames": 10, + "persistent_camera": true, + "device_resident_rate_and_output": true, + "include_truth": true, + "rng": "one generator seeded at camera construction and advanced per frame", + "cuda_synchronization": "before and after each timed region", + "construction_included": false, + "host_transfers_included": false + }, + "results": [ + { + "workflow": "pyramid_cmos_80", + "label": "Pyramid WFS CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 80, + 80 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "cpu", + "frames": 4295, + "elapsed_s": 2.0004646239976864, + "frame_s": 0.000465765919440672, + "frames_per_s": 2147.001225853703, + "megapixels_per_s": 13.740807845463701 + }, + { + "workflow": "pyramid_cmos_80", + "label": "Pyramid WFS CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 80, + 80 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 5011, + "elapsed_s": 2.0001363799965475, + "frame_s": 0.0003991491478739867, + "frames_per_s": 2505.329161608795, + "megapixels_per_s": 16.034106634296286 + }, + { + "workflow": "shack_hartmann_cmos_160", + "label": "Shack-Hartmann WFS CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 160, + 160 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "cpu", + "frames": 1149, + "elapsed_s": 2.001656759006437, + "frame_s": 0.001742085952137891, + "frames_per_s": 574.0244898782395, + "megapixels_per_s": 14.69502694088293 + }, + { + "workflow": "shack_hartmann_cmos_160", + "label": "Shack-Hartmann WFS CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 160, + 160 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 5026, + "elapsed_s": 2.000239181012148, + "frame_s": 0.0003979783487887282, + "frames_per_s": 2512.699504994586, + "megapixels_per_s": 64.3251073278614 + }, + { + "workflow": "ocam2k_emccd_240", + "label": "OCAM2K EMCCD", + "preset": "andor_ocam2k", + "sensor": "EMCCD", + "shape": [ + 240, + 240 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "cpu", + "frames": 288, + "elapsed_s": 2.003861945006065, + "frame_s": 0.006957853975715504, + "frames_per_s": 143.72247585106385, + "megapixels_per_s": 8.278414609021278 + }, + { + "workflow": "ocam2k_emccd_240", + "label": "OCAM2K EMCCD", + "preset": "andor_ocam2k", + "sensor": "EMCCD", + "shape": [ + 240, + 240 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 3420, + "elapsed_s": 2.0004429029941093, + "frame_s": 0.000584924825436874, + "frames_per_s": 1709.621401781179, + "megapixels_per_s": 98.47419274259589 + }, + { + "workflow": "saphira_eapd_256x320", + "label": "SAPHIRA eAPD", + "preset": "leonardo_saphira", + "sensor": "EAPD", + "shape": [ + 256, + 320 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "cpu", + "frames": 228, + "elapsed_s": 2.0011549520131666, + "frame_s": 0.00877699540356652, + "frames_per_s": 113.93420572986187, + "megapixels_per_s": 9.333490133390285 + }, + { + "workflow": "saphira_eapd_256x320", + "label": "SAPHIRA eAPD", + "preset": "leonardo_saphira", + "sensor": "EAPD", + "shape": [ + 256, + 320 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 3786, + "elapsed_s": 2.000311462994432, + "frame_s": 0.000528344284995888, + "frames_per_s": 1892.7052461782241, + "megapixels_per_s": 155.05041376692012 + }, + { + "workflow": "science_cmos_1024", + "label": "Large science CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 1024, + 1024 + ], + "precision": "float32", + "exposure_s": 5.0, + "photon_rate_per_s": 200.0, + "device": "cpu", + "frames": 27, + "elapsed_s": 2.000652825983707, + "frame_s": 0.07409825281421137, + "frames_per_s": 13.495594862504088, + "megapixels_per_s": 14.151156878545088 + }, + { + "workflow": "science_cmos_1024", + "label": "Large science CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 1024, + 1024 + ], + "precision": "float32", + "exposure_s": 5.0, + "photon_rate_per_s": 200.0, + "device": "gpu", + "frames": 525, + "elapsed_s": 2.2030622210004367, + "frame_s": 0.0041963089923817845, + "frames_per_s": 238.30466293484497, + "megapixels_per_s": 249.880550241568 + } + ] +} diff --git a/benchmarks/device-results-neoverse-n1-rtxa400.json b/benchmarks/device-results-neoverse-n1-rtxa400.json new file mode 100644 index 0000000..5555cbf --- /dev/null +++ b/benchmarks/device-results-neoverse-n1-rtxa400.json @@ -0,0 +1,126 @@ +{ + "schema_version": 1, + "generated_at_utc": "2026-10-07T07:14:37.452446+00:00", + "command": "/home/jtaylor/miniforge3/envs/aosim-bench/bin/python benchmarks/bench_devices.py --seconds 2 --warmup 10 --device gpu --output /home/jtaylor/aosim-bench-logs/artifacts/getframes-device-rtxa400.json", + "revision": "7b01daf765d432ab141176548f2b11cf0e7fc11c", + "source_dirty": true, + "python": "3.13.15 | packaged by conda-forge | (main, Sep 2 2026, 22:02:03) [GCC 15.3.0]", + "platform": "Linux-6.17.9-76061709-generic-aarch64-with-glibc2.39", + "processor": "aarch64", + "cpu": "Neoverse-N1", + "gpu": "NVIDIA RTX A400", + "dependencies": { + "getframes": "2.4.0", + "numpy": "2.5.3", + "scipy": "1.18.1", + "cupy": "14.2.0" + }, + "methodology": { + "seconds_per_cell": 2.0, + "warmup_frames": 10, + "persistent_camera": true, + "device_resident_rate_and_output": true, + "include_truth": true, + "rng": "one generator seeded at camera construction and advanced per frame", + "cuda_synchronization": "before and after each timed region", + "construction_included": false, + "host_transfers_included": false + }, + "results": [ + { + "workflow": "pyramid_cmos_80", + "label": "Pyramid WFS CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 80, + 80 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 4672, + "elapsed_s": 2.0003948630183004, + "frame_s": 0.0004281667086939855, + "frames_per_s": 2335.5388910320644, + "megapixels_per_s": 14.947448902605213 + }, + { + "workflow": "shack_hartmann_cmos_160", + "label": "Shack-Hartmann WFS CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 160, + 160 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 3458, + "elapsed_s": 2.0286170829785988, + "frame_s": 0.00058664461624598, + "frames_per_s": 1704.6095239041624, + "megapixels_per_s": 43.638003811946554 + }, + { + "workflow": "ocam2k_emccd_240", + "label": "OCAM2K EMCCD", + "preset": "andor_ocam2k", + "sensor": "EMCCD", + "shape": [ + 240, + 240 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 1073, + "elapsed_s": 2.0662656150234398, + "frame_s": 0.0019256902283536252, + "frames_per_s": 519.2943212133101, + "megapixels_per_s": 29.911352901886666 + }, + { + "workflow": "saphira_eapd_256x320", + "label": "SAPHIRA eAPD", + "preset": "leonardo_saphira", + "sensor": "EAPD", + "shape": [ + 256, + 320 + ], + "precision": "float32", + "exposure_s": 0.001, + "photon_rate_per_s": 2000000.0, + "device": "gpu", + "frames": 807, + "elapsed_s": 2.099866060016211, + "frame_s": 0.0026020645105529257, + "frames_per_s": 384.3102259549687, + "megapixels_per_s": 31.482693710231036 + }, + { + "workflow": "science_cmos_1024", + "label": "Large science CMOS", + "preset": "generic_cmos", + "sensor": "CMOS", + "shape": [ + 1024, + 1024 + ], + "precision": "float32", + "exposure_s": 5.0, + "photon_rate_per_s": 200.0, + "device": "gpu", + "frames": 133, + "elapsed_s": 3.1430114879913162, + "frame_s": 0.02363166532324298, + "frames_per_s": 42.31610368214074, + "megapixels_per_s": 44.37165073460441 + } + ] +} diff --git a/benchmarks/device-results-neoverse-n1.md b/benchmarks/device-results-neoverse-n1.md new file mode 100644 index 0000000..0217ae2 --- /dev/null +++ b/benchmarks/device-results-neoverse-n1.md @@ -0,0 +1,33 @@ +# CPU/GPU detector throughput on an Arm host + +Frames/s, higher is better. These come from +`benchmarks/device-results-neoverse-n1-rtx4060.json` and +`benchmarks/device-results-neoverse-n1-rtxa400.json`. They use the same +method and workflows as [device-results.md](device-results.md), whose +x86 + RTX 5090 numbers (getframes 2.1.1) are repeated here for reference. + +- Host: cfl-test-bench, an 80-core Ampere Neoverse-N1 (aarch64), shared. + Every run was pinned to 16 cores (`taskset -c 16-31`) with 16 BLAS + threads. +- GPUs: NVIDIA GeForce RTX 4060 (8 GB) and NVIDIA RTX A400 (4 GB), driver 580. +- Dependencies: getframes 2.4.0, NumPy 2.5.3, SciPy 1.18.1, CuPy 14.2.0. +- Method: persistent float32 camera, warm device-resident rate and output, + truth enabled, construction and host transfers excluded, CUDA synchronized. + +| Workflow | Detector | Native shape | Neoverse-N1 CPU (16 cores) | RTX A400 | RTX 4060 | Ryzen 9 9950X3D CPU | RTX 5090 | +| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Pyramid WFS CMOS | CMOS | 80x80 | 2,147.0 | 2,335.5 | 2,505.3 | 5,240.3 | 11,513.6 | +| Shack-Hartmann WFS CMOS | CMOS | 160x160 | 574.0 | 1,704.6 | 2,512.7 | 1,386.3 | 11,470.8 | +| OCAM2K EMCCD | EMCCD | 240x240 | 143.7 | 519.3 | 1,709.6 | 357.0 | 8,045.0 | +| SAPHIRA eAPD | EAPD | 256x320 | 113.9 | 384.3 | 1,892.7 | 280.4 | 7,496.5 | +| Large science CMOS | CMOS | 1024x1024 | 13.5 | 42.3 | 238.3 | 30.8 | 1,453.2 | + +Reading this: + +- **The GPU advantage grows with the detector.** At 80x80 the GPU barely + beats 16 N1 cores (1.2x on the RTX 4060), because each frame is a handful + of kernel launches. At 1024x1024 it is 18x. +- **The two cards only separate on large detectors.** The RTX 4060 runs + 1.1x the RTX A400 at 80x80, 3.3x on the OCAM2K and 5.6x at 1024x1024. +- **16 Neoverse-N1 cores reach a steady 0.40–0.44x a 16-core Ryzen 9 + 9950X3D** across every workflow. diff --git a/docs/guides/gpu.md b/docs/guides/gpu.md index a76232b..dddf9db 100644 --- a/docs/guides/gpu.md +++ b/docs/guides/gpu.md @@ -119,6 +119,8 @@ for every cell. The checked-in snapshot was intentionally recorded from the GPU development checkout, so it is evidence rather than a release guarantee. See the [rendered snapshot](https://github.com/jacotay7/getframes/blob/main/benchmarks/device-results.md) and [raw JSON](https://github.com/jacotay7/getframes/blob/main/benchmarks/device-results.json). +An Arm data point (Ampere Neoverse-N1, 16 pinned cores, RTX 4060 and RTX A400) +is in [device-results-neoverse-n1.md](https://github.com/jacotay7/getframes/blob/main/benchmarks/device-results-neoverse-n1.md). An additional owner-isolation benchmark covers structured-detector digitization. On this repository's Quadro P620, native float32 amplifier and bias maps reduced