From 5d0fb83678e38f3e675d70f124b7740b6239225e Mon Sep 17 00:00:00 2001 From: Dan LaManna Date: Tue, 6 Oct 2026 16:58:32 -0400 Subject: [PATCH 1/2] Add a workflow that backfills benchmark history --- .github/workflows/benchmarks-backfill.yml | 115 ++++++++++++++++++++++ benchmarks/README.md | 1 + benchmarks/history.py | 6 +- 3 files changed, 121 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/benchmarks-backfill.yml diff --git a/.github/workflows/benchmarks-backfill.yml b/.github/workflows/benchmarks-backfill.yml new file mode 100644 index 00000000..9bbd5581 --- /dev/null +++ b/.github/workflows/benchmarks-backfill.yml @@ -0,0 +1,115 @@ +name: benchmarks-backfill +on: + workflow_dispatch: + inputs: + since: + description: Benchmark each master merge since this date (YYYY-MM-DD) that has no results + required: true +permissions: + # The job pushes its results to the gh-pages branch. + contents: write +jobs: + backfill: + # One runner benchmarks every commit, so the differences between them come from the code rather + # than the hardware. Roughly 30 commits fit in the 6 hour job limit. + runs-on: ubuntu-24.04 + # The same as benchmarks.yml. + services: + postgres: + image: pgvector/pgvector:0.8.6-pg18 + env: + POSTGRES_PASSWORD: postgres + options: >- + --health-cmd "pg_isready --username postgres" + --health-start-period 30s + --health-start-interval 2s + ports: + - 5432:5432 + elasticsearch: + image: elasticsearch:9.5.2 + env: + ES_JAVA_OPTS: "-Xms250m -Xmx750m" + discovery.type: single-node + xpack.security.enabled: "true" + ELASTIC_PASSWORD: elastic + options: >- + --health-cmd "curl --fail --user elastic:elastic http://localhost:9200/" + --health-start-period 30s + --health-start-interval 2s + ports: + - 9200:9200 + env: + DJANGO_DATABASE_URL: postgres://postgres:postgres@localhost:5432/django + DJANGO_ISIC_ELASTICSEARCH_URL: http://elastic:elastic@localhost:9200 + DJANGO_CELERY_BROKER_URL: amqp://localhost:5672/ + DJANGO_CACHE_URL: redis://localhost:6379/0 + steps: + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 0 + - name: Install uv + uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2 + - name: Increase maintenance_work_mem for pgvector + run: psql -c "ALTER SYSTEM SET maintenance_work_mem = '512MB'" -c "SELECT pg_reload_conf();" + env: + PGHOST: localhost + PGUSER: postgres + PGPASSWORD: postgres + - name: Benchmark commits without results + # Each commit runs the current benchmarks on its own locked dependencies, the way tox's + # benchmark environment does. A commit can predate both. + run: | + benchmark_group=$(git show "$GITHUB_SHA:pyproject.toml" \ + | python3 -c "import sys, tomllib; print(*tomllib.load(sys.stdin.buffer)['dependency-groups']['benchmark'], sep=',')") + # Without a time, git uses the current time of day. + shas=$(git log --first-parent --since="$SINCE 00:00" --format=%H | jq --raw-input --null-input --raw-output \ + --rawfile data <(git show origin/gh-pages:benchmarks/data.js) \ + '[inputs] - [$data | ltrimstr("window.BENCHMARK_DATA = ") | fromjson | .entries.Benchmark[].commit.id] | .[]') + if [ -z "$shas" ]; then + echo "::error::Every master merge since $SINCE has results." + exit 1 + fi + + mkdir "$RUNNER_TEMP/reports" + for sha in $shas; do + git checkout --quiet --force "$sha" + git checkout "$GITHUB_SHA" -- benchmarks isic/settings/benchmark.py + uv run --locked --extra development --no-default-groups --group test --with "$benchmark_group" \ + pytest benchmarks --ds=isic.settings.benchmark --no-cov \ + --benchmark-disable-gc --benchmark-group-by=func --benchmark-columns=min,median,iqr,rounds \ + --benchmark-json="$RUNNER_TEMP/reports/$sha.json" \ + || echo "$sha" >> "$RUNNER_TEMP/failed" + done + git checkout --quiet --force "$GITHUB_SHA" + env: + SINCE: ${{ inputs.since }} + PYTHONPATH: ${{ github.workspace }} + - name: Checkout gh-pages + # After the benchmarks, so it has the results benchmarks.yml pushed meanwhile. + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: gh-pages + path: gh-pages + - name: Record results + run: | + for report in "$RUNNER_TEMP"/reports/*.json; do + python3 benchmarks/history.py record "$report" gh-pages/benchmarks/data.js + done + - name: Push results + working-directory: gh-pages + run: | + # See benchmarks.yml. + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add benchmarks + git commit --message "Record backfilled benchmarks since $SINCE" + git push + env: + SINCE: ${{ inputs.since }} + - name: Check for failed commits + run: | + if [ -s "$RUNNER_TEMP/failed" ]; then + echo "::error::The benchmarks failed for $(paste -sd ' ' "$RUNNER_TEMP/failed")." + exit 1 + fi diff --git a/benchmarks/README.md b/benchmarks/README.md index b7e40150..2ff2239f 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -17,6 +17,7 @@ uv run tox -e benchmark -- --benchmark-disable # run each benchmark once, w - Benchmarks that write data roll back each round. - Benchmarks that use the `benchmark_with_memory` fixture also record the peak memory Python allocates, measured in a separate call. Memory allocated outside Python, such as by psycopg, isn't counted. - CI (`.github/workflows/benchmarks.yml`) runs on each push to master. `history.py record` adds each benchmark's fastest round and peak memory to `benchmarks/data.js` on the `gh-pages` branch, and the job copies `index.html` next to it. Then `history.py check` fails the job when either is more than 50% higher than in the run before it. +- `.github/workflows/benchmarks-backfill.yml` benchmarks the master merges since a date that have no results yet, such as ones from before the benchmarks existed or whose run failed, with the current benchmarks. It runs them all on one runner, so roughly 30 fit in the 6 hour job limit. For more, run it with a later date first, then an earlier one. Run it with `gh workflow run benchmarks-backfill.yml -f since=YYYY-MM-DD`. - `index.html` charts `data.js`, grouped by module and test, at https://imagemarkup.github.io/isic/benchmarks/. To chart local runs: ```sh diff --git a/benchmarks/history.py b/benchmarks/history.py index f72a56b9..9a845877 100644 --- a/benchmarks/history.py +++ b/benchmarks/history.py @@ -8,6 +8,7 @@ the peak memory of the benchmarks that record it. """ +from datetime import datetime import json from pathlib import Path import subprocess @@ -39,7 +40,8 @@ def record(report_path: Path, data_path: Path) -> None: data = load(data_path) data["lastUpdate"] = time.time() * 1000 - data["entries"]["Benchmark"].append( + runs = data["entries"]["Benchmark"] + runs.append( { "commit": { "id": commit["id"], @@ -66,6 +68,8 @@ def record(report_path: Path, data_path: Path) -> None: ], } ) + # Backfilled runs are recorded after newer ones. + runs.sort(key=lambda run: datetime.fromisoformat(run["commit"]["timestamp"])) data_path.parent.mkdir(parents=True, exist_ok=True) data_path.write_text(PREFIX + json.dumps(data)) From e8fa6bd846048f119532d51ea873853ef7c40fea Mon Sep 17 00:00:00 2001 From: Dan LaManna Date: Tue, 6 Oct 2026 17:52:00 -0400 Subject: [PATCH 2/2] Run backfilled benchmarks without write access --- .github/workflows/benchmarks-backfill.yml | 48 +++++++++++++++++------ 1 file changed, 37 insertions(+), 11 deletions(-) diff --git a/.github/workflows/benchmarks-backfill.yml b/.github/workflows/benchmarks-backfill.yml index 9bbd5581..f5e396cf 100644 --- a/.github/workflows/benchmarks-backfill.yml +++ b/.github/workflows/benchmarks-backfill.yml @@ -6,10 +6,9 @@ on: description: Benchmark each master merge since this date (YYYY-MM-DD) that has no results required: true permissions: - # The job pushes its results to the gh-pages branch. - contents: write + contents: read jobs: - backfill: + benchmark: # One runner benchmarks every commit, so the differences between them come from the code rather # than the hardware. Roughly 30 commits fit in the 6 hour job limit. runs-on: ubuntu-24.04 @@ -48,6 +47,8 @@ jobs: uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 + # This job runs each commit's code and dependencies, so it keeps no credentials. + persist-credentials: false - name: Install uv uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2 - name: Increase maintenance_work_mem for pgvector @@ -85,15 +86,46 @@ jobs: env: SINCE: ${{ inputs.since }} PYTHONPATH: ${{ github.workspace }} + - name: Upload results + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: reports + path: ${{ runner.temp }}/reports + if-no-files-found: error + - name: Check for failed commits + run: | + if [ -s "$RUNNER_TEMP/failed" ]; then + echo "::error::The benchmarks failed for $(paste -sd ' ' "$RUNNER_TEMP/failed")." + exit 1 + fi + publish: + needs: benchmark + # Record the commits that worked, even when others failed. + if: ${{ !cancelled() }} + runs-on: ubuntu-24.04 + permissions: + # The job pushes the results to the gh-pages branch. + contents: write + steps: + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # history.py reads each commit's message. + fetch-depth: 0 + persist-credentials: false - name: Checkout gh-pages - # After the benchmarks, so it has the results benchmarks.yml pushed meanwhile. uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: ref: gh-pages path: gh-pages + - name: Download results + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: reports + path: reports - name: Record results run: | - for report in "$RUNNER_TEMP"/reports/*.json; do + for report in reports/*.json; do python3 benchmarks/history.py record "$report" gh-pages/benchmarks/data.js done - name: Push results @@ -107,9 +139,3 @@ jobs: git push env: SINCE: ${{ inputs.since }} - - name: Check for failed commits - run: | - if [ -s "$RUNNER_TEMP/failed" ]; then - echo "::error::The benchmarks failed for $(paste -sd ' ' "$RUNNER_TEMP/failed")." - exit 1 - fi