From 91009d4b0cf25218f1f737cf2855bf5b016f643c Mon Sep 17 00:00:00 2001 From: Shaun Eccles Date: Sun, 27 Sep 2026 14:56:40 +1000 Subject: [PATCH] test: skip timing benchmarks in CI The threading/asyncio/datatype performance tests measure wall-clock speedups, which shared CI runners can't deliver reliably: the large-data threading test failed at 0.88x on macos-15-intel and 0.68x on windows-latest with no code change. Mark every timing test `perf` and deselect them in all CI pytest runs (-m "not perf"). The correctness tests in those files (GIL release output quality, release_gil parameter) keep running. Locally, plain `pytest` still runs everything. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/pythonpackage.yml | 6 +++--- pyproject.toml | 7 ++++++- tests/test_asyncio_performance.py | 4 ++++ tests/test_datatype_performance.py | 5 +++++ tests/test_threading_performance.py | 6 ++++++ 5 files changed, 24 insertions(+), 4 deletions(-) diff --git a/.github/workflows/pythonpackage.yml b/.github/workflows/pythonpackage.yml index ae77b21..1749329 100644 --- a/.github/workflows/pythonpackage.yml +++ b/.github/workflows/pythonpackage.yml @@ -82,7 +82,7 @@ jobs: run: | uv venv uv pip install dist/*.whl --group test - uv run --no-project pytest tests + uv run --no-project pytest tests -m "not perf" - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: @@ -118,7 +118,7 @@ jobs: - name: Test run: | uv run --no-project python -c "import numpy; print('numpy', numpy.__version__)" - uv run --no-project pytest tests + uv run --no-project pytest tests -m "not perf" # Early warning: NumPy's nightly builds on the newest CPython. Not a release # gate; it only runs on the schedule or by hand. @@ -151,7 +151,7 @@ jobs: - name: Test run: | uv run --no-sync python -c "import numpy; print('numpy', numpy.__version__)" - uv run --no-sync pytest tests + uv run --no-sync pytest tests -m "not perf" publish: name: Publish to PyPI diff --git a/pyproject.toml b/pyproject.toml index 857c085..7c37eba 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -48,11 +48,16 @@ test = [ [tool.setuptools.dynamic] readme = {file = "README.md", content-type = "text/markdown"} +[tool.pytest.ini_options] +markers = [ + "perf: timing benchmarks; flaky on shared CI runners, so CI deselects them with -m 'not perf'", +] + [tool.setuptools_scm] [tool.cibuildwheel] test-groups = ["test"] -test-command = "pytest {project}/tests" +test-command = "pytest {project}/tests -m \"not perf\"" build-frontend = "build[uv]" build = ["cp311-*", "cp312-*", "cp313-*", "cp314-*"] # Skip 32-bit builds and musllinux wheels diff --git a/tests/test_asyncio_performance.py b/tests/test_asyncio_performance.py index 72e2480..34744b6 100644 --- a/tests/test_asyncio_performance.py +++ b/tests/test_asyncio_performance.py @@ -23,6 +23,10 @@ import samplerate +# Timing benchmarks: shared CI runners make them flaky, so CI runs +# pytest -m "not perf". Run them locally with plain pytest. +pytestmark = pytest.mark.perf + def is_arm_mac(): """Check if running on ARM-based macOS (Apple Silicon).""" diff --git a/tests/test_datatype_performance.py b/tests/test_datatype_performance.py index de14797..7fa79ea 100644 --- a/tests/test_datatype_performance.py +++ b/tests/test_datatype_performance.py @@ -1,7 +1,12 @@ import time import numpy as np +import pytest import samplerate +# Timing benchmarks: shared CI runners make them flaky, so CI runs +# pytest -m "not perf". Run them locally with plain pytest. +pytestmark = pytest.mark.perf + def benchmark_resample(input_data, ratio=1.5, converter='sinc_fastest'): start_time = time.perf_counter() samplerate.resample(input_data, ratio, converter) diff --git a/tests/test_threading_performance.py b/tests/test_threading_performance.py index c535ffb..fb1dae9 100644 --- a/tests/test_threading_performance.py +++ b/tests/test_threading_performance.py @@ -55,6 +55,7 @@ def producer(): return output +@pytest.mark.perf @pytest.mark.parametrize("num_threads", [2, 4, 6, 8]) @pytest.mark.parametrize("converter_type", ["sinc_fastest", "sinc_medium", "sinc_best"]) def test_resample_gil_release_parallel(num_threads, converter_type): @@ -119,6 +120,7 @@ def test_resample_gil_release_parallel(num_threads, converter_type): print(f" ✓ Performance meets expectations ({expected_speedup}x)") +@pytest.mark.perf @pytest.mark.parametrize("num_threads", [2, 4, 6, 8]) @pytest.mark.parametrize("converter_type", ["sinc_fastest", "sinc_medium", "sinc_best"]) def test_resampler_process_gil_release_parallel(num_threads, converter_type): @@ -175,6 +177,7 @@ def test_resampler_process_gil_release_parallel(num_threads, converter_type): print(f" ✓ Performance meets expectations ({expected_speedup}x)") +@pytest.mark.perf @pytest.mark.parametrize("num_threads", [2, 4, 6, 8]) @pytest.mark.parametrize("converter_type", ["sinc_fastest", "sinc_medium", "sinc_best"]) def test_callback_resampler_gil_release_parallel(num_threads, converter_type): @@ -275,6 +278,7 @@ def worker(data, ratio, results, index): assert np.allclose(results[0], results[1]) +@pytest.mark.perf def test_conditional_gil_release_small_data(): """Test that small data sizes perform well without GIL release overhead. @@ -310,6 +314,7 @@ def test_conditional_gil_release_small_data(): assert per_call_us > 0 +@pytest.mark.perf def test_conditional_gil_release_large_data_threading(): """Test that large data sizes still benefit from GIL release for threading. @@ -430,6 +435,7 @@ def test_release_gil_parameter_invalid(): print("\n Invalid release_gil parameter test passed!") +@pytest.mark.perf def test_gil_metrics_report(): """Generate a detailed performance report for GIL release optimization.""" print("\n" + "="*70)