diff --git a/.github/workflows/benchmarks-backfill.yml b/.github/workflows/benchmarks-backfill.yml new file mode 100644 index 00000000..f5e396cf --- /dev/null +++ b/.github/workflows/benchmarks-backfill.yml @@ -0,0 +1,141 @@ +name: benchmarks-backfill +on: + workflow_dispatch: + inputs: + since: + description: Benchmark each master merge since this date (YYYY-MM-DD) that has no results + required: true +permissions: + contents: read +jobs: + benchmark: + # One runner benchmarks every commit, so the differences between them come from the code rather + # than the hardware. Roughly 30 commits fit in the 6 hour job limit. + runs-on: ubuntu-24.04 + # The same as benchmarks.yml. + services: + postgres: + image: pgvector/pgvector:0.8.6-pg18 + env: + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd "pg_isready --username postgres" + --health-start-period 30s + --health-start-interval 2s + ports: + - 5432:5432 + elasticsearch: + image: elasticsearch:9.5.2 + env: + ES_JAVA_OPTS: "-Xms250m -Xmx750m" + discovery.type: single-node + xpack.security.enabled: "true" + ELASTIC_PASSWORD: elastic + options: >- + --health-cmd "curl --fail --user elastic:elastic http://localhost:9200/" + --health-start-period 30s + --health-start-interval 2s + ports: + - 9200:9200 + env: + DJANGO_DATABASE_URL: postgres://postgres:postgres@localhost:5432/django + DJANGO_ISIC_ELASTICSEARCH_URL: http://elastic:elastic@localhost:9200 + DJANGO_CELERY_BROKER_URL: amqp://localhost:5672/ + DJANGO_CACHE_URL: redis://localhost:6379/0 + steps: + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 0 + # This job runs each commit's code and dependencies, so it keeps no credentials. + persist-credentials: false + - name: Install uv + uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2 + - name: Increase maintenance_work_mem for pgvector + run: psql -c "ALTER SYSTEM SET maintenance_work_mem = '512MB'" -c "SELECT pg_reload_conf();" + env: + PGHOST: localhost + PGUSER: postgres + PGPASSWORD: postgres + - name: Benchmark commits without results + # Each commit runs the current benchmarks on its own locked dependencies, the way tox's + # benchmark environment does. A commit can predate both. + run: | + benchmark_group=$(git show "$GITHUB_SHA:pyproject.toml" \ + | python3 -c "import sys, tomllib; print(*tomllib.load(sys.stdin.buffer)['dependency-groups']['benchmark'], sep=',')") + # Without a time, git uses the current time of day. + shas=$(git log --first-parent --since="$SINCE 00:00" --format=%H | jq --raw-input --null-input --raw-output \ + --rawfile data <(git show origin/gh-pages:benchmarks/data.js) \ + '[inputs] - [$data | ltrimstr("window.BENCHMARK_DATA = ") | fromjson | .entries.Benchmark[].commit.id] | .[]') + if [ -z "$shas" ]; then + echo "::error::Every master merge since $SINCE has results." + exit 1 + fi + + mkdir "$RUNNER_TEMP/reports" + for sha in $shas; do + git checkout --quiet --force "$sha" + git checkout "$GITHUB_SHA" -- benchmarks isic/settings/benchmark.py + uv run --locked --extra development --no-default-groups --group test --with "$benchmark_group" \ + pytest benchmarks --ds=isic.settings.benchmark --no-cov \ + --benchmark-disable-gc --benchmark-group-by=func --benchmark-columns=min,median,iqr,rounds \ + --benchmark-json="$RUNNER_TEMP/reports/$sha.json" \ + || echo "$sha" >> "$RUNNER_TEMP/failed" + done + git checkout --quiet --force "$GITHUB_SHA" + env: + SINCE: ${{ inputs.since }} + PYTHONPATH: ${{ github.workspace }} + - name: Upload results + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: reports + path: ${{ runner.temp }}/reports + if-no-files-found: error + - name: Check for failed commits + run: | + if [ -s "$RUNNER_TEMP/failed" ]; then + echo "::error::The benchmarks failed for $(paste -sd ' ' "$RUNNER_TEMP/failed")." + exit 1 + fi + publish: + needs: benchmark + # Record the commits that worked, even when others failed. + if: ${{ !cancelled() }} + runs-on: ubuntu-24.04 + permissions: + # The job pushes the results to the gh-pages branch. + contents: write + steps: + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # history.py reads each commit's message. + fetch-depth: 0 + persist-credentials: false + - name: Checkout gh-pages + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: gh-pages + path: gh-pages + - name: Download results + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: reports + path: reports + - name: Record results + run: | + for report in reports/*.json; do + python3 benchmarks/history.py record "$report" gh-pages/benchmarks/data.js + done + - name: Push results + working-directory: gh-pages + run: | + # See benchmarks.yml. + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add benchmarks + git commit --message "Record backfilled benchmarks since $SINCE" + git push + env: + SINCE: ${{ inputs.since }} diff --git a/benchmarks/README.md b/benchmarks/README.md index b7e40150..2ff2239f 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -17,6 +17,7 @@ uv run tox -e benchmark -- --benchmark-disable # run each benchmark once, w - Benchmarks that write data roll back each round. - Benchmarks that use the `benchmark_with_memory` fixture also record the peak memory Python allocates, measured in a separate call. Memory allocated outside Python, such as by psycopg, isn't counted. - CI (`.github/workflows/benchmarks.yml`) runs on each push to master. `history.py record` adds each benchmark's fastest round and peak memory to `benchmarks/data.js` on the `gh-pages` branch, and the job copies `index.html` next to it. Then `history.py check` fails the job when either is more than 50% higher than in the run before it. +- `.github/workflows/benchmarks-backfill.yml` benchmarks the master merges since a date that have no results yet, such as ones from before the benchmarks existed or whose run failed, with the current benchmarks. It runs them all on one runner, so roughly 30 fit in the 6 hour job limit. For more, run it with a later date first, then an earlier one. Run it with `gh workflow run benchmarks-backfill.yml -f since=YYYY-MM-DD`. - `index.html` charts `data.js`, grouped by module and test, at https://imagemarkup.github.io/isic/benchmarks/. To chart local runs: ```sh diff --git a/benchmarks/history.py b/benchmarks/history.py index f72a56b9..9a845877 100644 --- a/benchmarks/history.py +++ b/benchmarks/history.py @@ -8,6 +8,7 @@ the peak memory of the benchmarks that record it. """ +from datetime import datetime import json from pathlib import Path import subprocess @@ -39,7 +40,8 @@ def record(report_path: Path, data_path: Path) -> None: data = load(data_path) data["lastUpdate"] = time.time() * 1000 - data["entries"]["Benchmark"].append( + runs = data["entries"]["Benchmark"] + runs.append( { "commit": { "id": commit["id"], @@ -66,6 +68,8 @@ def record(report_path: Path, data_path: Path) -> None: ], } ) + # Backfilled runs are recorded after newer ones. + runs.sort(key=lambda run: datetime.fromisoformat(run["commit"]["timestamp"])) data_path.parent.mkdir(parents=True, exist_ok=True) data_path.write_text(PREFIX + json.dumps(data))