From 6ba2c7ac4200c3d00e5725497ae37a9e74bcfe07 Mon Sep 17 00:00:00 2001 From: Steve Gerbino Date: Tue, 29 Sep 2026 21:47:38 +0200 Subject: [PATCH] ci: run performance benchmarks on PRs with a per-platform verdict MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a benchmarks workflow that builds the PR head and its merge-base, runs the full bench suite in interleaved base/head passes on dedicated self-hosted runners (Linux epoll+uring, Windows iocp, macOS kqueue), and posts one advisory PR comment, updated in place on each push. A row is flagged only when the median of its paired per-iteration deltas exceeds the larger of a fixed minimum effect size and three times the sample spread of the noisier side, so the noise floor is measured live on the same machine rather than maintained as stored calibration. Iterations alternate starting side so linear drift cancels. A workflow_dispatch A/A mode benchmarks the merge-base against itself to audit that flag rule whenever the hardware or the suite changes. The comparison and report generation live in a stdlib-only script, .github/bench/compare.py. It matches benchmark suites dynamically: benchmarks present on only one side are reported as added or removed rather than compared, unrecognized metrics degrade to an unsupported list, and malformed input files are warned about and skipped — the report never fails a job over suite shape. Untrusted code never reaches the machines without review: the pull_request_target gate runs collaborator PRs automatically and everyone else's only when a maintainer applies the benchmark label, which is consumed on use and stripped on later pushes so each approval covers exactly the diff that was reviewed. Head and base SHAs are pinned at gate time, and all tooling (this script and the composite actions) executes from the trusted base commit, never from the pull request. --- .github/bench/compare.py | 383 +++++++++++++++++++++++++++++++ .github/workflows/benchmarks.yml | 381 ++++++++++++++++++++++++++++++ 2 files changed, 764 insertions(+) create mode 100644 .github/bench/compare.py create mode 100644 .github/workflows/benchmarks.yml diff --git a/.github/bench/compare.py b/.github/bench/compare.py new file mode 100644 index 000000000..bb526503d --- /dev/null +++ b/.github/bench/compare.py @@ -0,0 +1,383 @@ +#!/usr/bin/env python3 +"""Compare interleaved corosio benchmark runs. + +summarize: reduce raw per-iteration JSON (base/head × backend) to one +per-platform summary with flagging. +report: merge per-platform summaries into the PR comment markdown. + +Input files: --.json as written by +corosio_bench --output. Never exits nonzero because of suite-shape +differences between base and head; only real I/O or usage errors fail. + +Per benchmark, delta_pct is the median of the per-iteration paired +deltas (resists a single outlier iteration) and noise_pct is the +larger of the base and head sample CVs (a shift confined to one side +still sets a floor). +""" +import argparse +import json +import re +import statistics +import sys +from pathlib import Path + +MIN_EFFECT_PCT = 2.0 +NOISE_FACTOR = 3.0 +HIGHER_BETTER = ("bytes_per_sec", "items_per_sec", "ops_per_sec") +FNAME = re.compile(r"^(base|head)-([A-Za-z0-9_]+)-(\d+)\.json$") + + +def primary_metric(category, metrics): + """Pick the compared metric and its direction for one benchmark.""" + if "latency" in category and "latency_mean_ns" in metrics: + return "latency_mean_ns", "lower" + for m in HIGHER_BETTER: + if m in metrics: + return m, "higher" + if "latency_mean_ns" in metrics: + return "latency_mean_ns", "lower" + return None, None + + +def human(value, metric): + """Format a metric value with readable units.""" + if metric.endswith("_ns"): + for factor, unit in ((1e9, "s"), (1e6, "ms"), (1e3, "µs")): + if abs(value) >= factor: + return f"{value / factor:.2f} {unit}" + return f"{value:.0f} ns" + # Bytes glue the prefix to the unit (KB/s, GB/s); count-style metrics + # glue it to the number (1.82K ops/s) so sub-1000 values still carry a + # unit instead of a bare "/s". + if metric == "bytes_per_sec": + for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")): + if abs(value) >= factor: + n = value / factor + return (f"{n:.2f} {prefix}B/s" if n < 100 + else f"{n:.1f} {prefix}B/s") + return f"{value:.1f} B/s" + unit = "items/s" if metric == "items_per_sec" else "ops/s" + for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")): + if abs(value) >= factor: + n = value / factor + return (f"{n:.2f}{prefix} {unit}" if n < 100 + else f"{n:.1f}{prefix} {unit}") + return f"{value:.1f} {unit}" + + +MARKER = "" +MAX_COMMENT_CHARS = 60000 + + +FLAGGED_HEADER = "| Platform | Backend | Benchmark | Metric | Base | Head | Δ | Noise |" +FLAGGED_RULE = "|---|---|---|---|---|---|---|---|" +DETAIL_HEADER = "| Benchmark | Backend | Metric | Base | Head | Δ | Noise |" +DETAIL_RULE = "|---|---|---|---|---|---|---|" + + +def _delta_cells(r): + # Deltas are sign-normalized (positive = improvement), so color tracks + # verdict on every row, flagged or not. No green arrow exists in emoji, + # so a colored dot carries the verdict and a text arrowhead the direction. + # Color follows the two-decimal value actually shown, so a delta that + # displays as 0.00% is neutral rather than a red "-0.00%". + shown = round(r["delta_pct"], 2) + if shown == 0: + delta = "⚪ 0.00%" + else: + arrow = "🔴▼" if shown < 0 else "🟢▲" + delta = f"{arrow} {shown:+.2f}%" + noise = "—" if r["noise_pct"] is None else f"{r['noise_pct']:.2f}%" + return (f"{human(r['base_mean'], r['metric'])} " + f"| {human(r['head_mean'], r['metric'])} " + f"| {delta} | {noise}") + + +def _flagged_line(platform, r): + return (f"| {platform} | {r['backend']} | {r['category']}/{r['name']} " + f"| {r['metric']} | {_delta_cells(r)} |") + + +def _detail_line(r): + return f"| {r['name']} | {r['backend']} | {r['metric']} | {_delta_cells(r)} |" + + +def _platform_detail_lines(platform, s, condensed): + """Full per-platform results table, or a one-line summary when condensed.""" + if condensed: + counts = [] + if s["new"]: + counts.append(f"{len(s['new'])} new") + if s["removed"]: + counts.append(f"{len(s['removed'])} removed") + if s["unsupported"]: + counts.append(f"{len(s['unsupported'])} unsupported") + suffix = f" ({', '.join(counts)})" if counts else "" + return [f"- **{platform}** — {len(s['rows'])} benchmarks, " + f"{s['iterations']} iterations, {s['duration_s']}s each" + f"{suffix}"] + + lines = [f"
{platform} — full results " + f"({len(s['rows'])} benchmarks, {s['iterations']} iterations, " + f"{s['duration_s']}s each)", ""] + # One table per category keeps rows narrow enough for GitHub's comment + # width; same-name rows sort adjacently so backends compare at a glance. + by_category = {} + for r in s["rows"]: + by_category.setdefault(r["category"], []).append(r) + for category in sorted(by_category): + lines += [f"**{category}**", "", DETAIL_HEADER, DETAIL_RULE] + rows = sorted(by_category[category], + key=lambda r: (r["name"], r["backend"])) + lines += [_detail_line(r) for r in rows] + lines.append("") + if s["new"]: + lines += ["**New benchmarks (no baseline):**", ""] + lines += [f"- `{n['category']}/{n['name']}` [{n['backend']}] " + f"{human(n['head_mean'], n['metric'])}" for n in s["new"]] + lines.append("") + if s["removed"]: + lines += ["**Removed benchmarks:** " + + ", ".join(f"`{r['category']}/{r['name']}`" + for r in s["removed"]), ""] + if s["unsupported"]: + lines += ["**Unsupported (no recognized metric):** " + + ", ".join(f"`{u['category']}/{u['name']}`" + for u in s["unsupported"]), ""] + lines += ["
", ""] + return lines + + +def _build_report(summaries, base_sha, head_sha, run_url, condensed): + lines = [MARKER, "## Benchmark report", ""] + mode = next((s["mode"] for s in summaries.values() if s), "ab") + if mode == "aa": + lines += ["**A/A validation run** — base compared against itself; " + "every flag below is a false positive.", ""] + lines += [f"`{base_sha[:12]}` (base) vs `{head_sha[:12]}` (head)", ""] + + for platform, s in summaries.items(): + if s is None: + lines.append(f"- ❌ **{platform}** — no results " + "(job failed or runner offline)") + elif s["flagged_count"]: + lines.append(f"- ⚠️ **{platform}** — {s['flagged_count']} flagged " + f"({', '.join(s['backends'])})") + else: + lines.append(f"- ✅ **{platform}** — clean " + f"({', '.join(s['backends'])})") + lines.append("") + + flagged = [(p, r) for p, s in summaries.items() if s + for r in s["rows"] if r["flagged"]] + if flagged: + lines += ["### ⚠️ Flagged", "", FLAGGED_HEADER, FLAGGED_RULE] + lines += [_flagged_line(p, r) for p, r in flagged] + lines.append("") + + for platform, s in summaries.items(): + if s is None: + continue + lines += _platform_detail_lines(platform, s, condensed) + + if condensed: + lines += ["_full tables omitted — comment size limit; " + "see the run artifacts_", ""] + + lines += [f"[Run & raw JSON artifacts]({run_url}) · " + "flag rule: |median Δ| > max(2%, 3×CV of the noisier side) " + "· advisory only"] + return "\n".join(lines) + "\n" + + +def report(summaries, base_sha, head_sha, run_url): + md = _build_report(summaries, base_sha, head_sha, run_url, condensed=False) + # GitHub caps issue comments at 65536 chars; a run with many benchmarks + # across three platforms can exceed that in full-details form. + if len(md) > MAX_COMMENT_CHARS: + md = _build_report(summaries, base_sha, head_sha, run_url, condensed=True) + return md + + +def load_runs(input_dir): + """Return {(side, backend, iter): {(category, name): {metric: value}}}.""" + runs = {} + for p in sorted(Path(input_dir).iterdir()): + m = FNAME.match(p.name) + if not m: + continue + side, backend, it = m.group(1), m.group(2), int(m.group(3)) + try: + payload = json.loads(p.read_text()) + except (OSError, json.JSONDecodeError) as e: + print(f"warning: skipping unreadable {p.name}: {e}", file=sys.stderr) + continue + if not isinstance(payload, dict): + print(f"warning: skipping non-object JSON {p.name}", file=sys.stderr) + continue + benchmarks = payload.get("benchmarks", []) + if not isinstance(benchmarks, list): + print(f"warning: skipping {p.name}: benchmarks field is not a list", file=sys.stderr) + continue + table = {} + for b in benchmarks: + if not isinstance(b, dict): + print(f"warning: skipping non-object benchmark entry in {p.name}", file=sys.stderr) + continue + key = (b.get("category", ""), b.get("name", "")) + table[key] = { + k: v for k, v in b.items() + if isinstance(v, (int, float)) and not isinstance(v, bool) + } + runs[(side, backend, it)] = table + return runs + + +def _values(runs, side, backend, key, metric): + """Metric samples for one benchmark on one side, ordered by iteration.""" + out = [] + for (s, b, it), table in sorted(runs.items(), key=lambda kv: kv[0][2]): + if s == side and b == backend and key in table and metric in table[key]: + out.append((it, table[key][metric])) + return out + + +def summarize(input_dir, platform, mode="ab"): + runs = load_runs(input_dir) + backends = sorted({b for (_, b, _) in runs}) + iterations = max((it for (_, _, it) in runs), default=0) + duration = 0.0 + rows, new, removed, unsupported = [], [], [], [] + + for backend in backends: + base_keys, head_keys = set(), set() + sample = {} + for (s, b, it), table in runs.items(): + if b != backend: + continue + (base_keys if s == "base" else head_keys).update(table) + for key, metrics in table.items(): + sample.setdefault(key, metrics) + + for key in sorted(base_keys | head_keys): + category, name = key + metric, direction = primary_metric(category, sample.get(key, {})) + if metric is None: + unsupported.append( + {"backend": backend, "category": category, "name": name}) + continue + if key not in base_keys: + head = _values(runs, "head", backend, key, metric) + mean = statistics.fmean(v for _, v in head) if head else 0.0 + new.append({"backend": backend, "category": category, + "name": name, "metric": metric, "head_mean": mean}) + continue + if key not in head_keys: + removed.append( + {"backend": backend, "category": category, "name": name}) + continue + + base = dict(_values(runs, "base", backend, key, metric)) + head = dict(_values(runs, "head", backend, key, metric)) + common = sorted(set(base) & set(head)) + deltas = [] + for it in common: + b_v, h_v = base[it], head[it] + if b_v == 0: + continue + d = (h_v - b_v) / b_v * 100.0 + if direction == "lower": + d = -d + deltas.append(d) + if not deltas: + unsupported.append( + {"backend": backend, "category": category, "name": name}) + continue + + base_vals = [base[it] for it in common] + head_vals = [head[it] for it in common] + base_mean = statistics.fmean(base_vals) + head_mean = statistics.fmean(head_vals) + + def cv(vals, mean): + if len(vals) >= 2 and mean != 0: + return statistics.stdev(vals) / abs(mean) * 100.0 + return None + + base_cv = cv(base_vals, base_mean) + head_cv = cv(head_vals, head_mean) + # a single-side outlier or a side-level shift can leave one + # side's spread tight while the other carries the noise + noise_candidates = [c for c in (base_cv, head_cv) if c is not None] + noise_pct = max(noise_candidates) if noise_candidates else None + # median resists a single blown-up iteration that a mean would not + delta_pct = statistics.median(deltas) + flagged = ( + noise_pct is not None + and abs(delta_pct) > max(MIN_EFFECT_PCT, NOISE_FACTOR * noise_pct) + ) + rows.append({ + "backend": backend, "category": category, "name": name, + "metric": metric, "direction": direction, + "base_mean": base_mean, "head_mean": head_mean, + "delta_pct": delta_pct, "noise_pct": noise_pct, + "flagged": flagged, + }) + + for p in Path(input_dir).iterdir(): + if FNAME.match(p.name): + try: + duration = json.loads(p.read_text())["metadata"]["duration_s"] + break + except Exception: + pass + + return { + "platform": platform, "backends": backends, + "iterations": iterations, "duration_s": duration, "mode": mode, + "rows": rows, "new": new, "removed": removed, + "unsupported": unsupported, + "flagged_count": sum(1 for r in rows if r["flagged"]), + } + + +def main(argv=None): + ap = argparse.ArgumentParser(prog="compare.py") + sub = ap.add_subparsers(dest="cmd", required=True) + s = sub.add_parser("summarize") + s.add_argument("--platform", required=True) + s.add_argument("--input-dir", required=True) + s.add_argument("--output", required=True) + s.add_argument("--mode", default="ab", choices=("ab", "aa")) + r = sub.add_parser("report") + r.add_argument("--summaries", required=True, + help="dir containing bench-/summary.json") + r.add_argument("--expect", required=True, + help="comma-separated platform list") + r.add_argument("--base-sha", required=True) + r.add_argument("--head-sha", required=True) + r.add_argument("--run-url", required=True) + r.add_argument("--output", required=True) + args = ap.parse_args(argv) + + if args.cmd == "summarize": + summary = summarize(args.input_dir, args.platform, args.mode) + Path(args.output).write_text(json.dumps(summary, indent=2)) + print(f"{args.platform}: {len(summary['rows'])} rows, " + f"{summary['flagged_count']} flagged") + elif args.cmd == "report": + summaries = {} + for platform in args.expect.split(","): + p = Path(args.summaries) / f"bench-{platform}" / "summary.json" + try: + summaries[platform] = json.loads(p.read_text()) + except (OSError, json.JSONDecodeError): + summaries[platform] = None + md = report(summaries, args.base_sha, args.head_sha, args.run_url) + Path(args.output).write_text(md) + print(f"report written: {args.output}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml new file mode 100644 index 000000000..51e87080a --- /dev/null +++ b/.github/workflows/benchmarks.yml @@ -0,0 +1,381 @@ +# +# Copyright (c) 2026 Steve Gerbino +# +# Distributed under the Boost Software License, Version 1.0. (See accompanying +# file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt) +# +# Official repository: https://github.com/cppalliance/corosio/ +# +# Per-PR performance benchmarks on dedicated self-hosted runners. +# Advisory only. See issue #343. +# +# Runner prerequisites (per machine, maintained by the infra admin): +# all : git, cmake >= 3.20, ninja or make/msbuild, python3 +# linux : gcc or clang, liburing-dev +# windows : MSVC (vcvars auto-detected by cmake), python3 on PATH, +# Git Bash (for shell: bash steps) +# macos : Xcode command line tools + +name: benchmarks + +on: + pull_request_target: + types: [opened, reopened, synchronize, labeled] + paths: + - 'include/**' + - 'src/**' + - 'bench/**' + - '.github/workflows/benchmarks.yml' + - '.github/bench/**' + workflow_dispatch: + inputs: + mode: + description: "ab = head vs merge-base, aa = base vs itself" + type: choice + options: [ab, aa] + default: ab + iterations: + description: "override BENCH_ITERATIONS" + default: "" + duration: + description: "override BENCH_DURATION" + default: "" + category: + description: "run only this benchmark category (default: all)" + default: "" + +concurrency: + group: benchmarks-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + gate: + runs-on: ubuntu-24.04 + permissions: + contents: read + pull-requests: write + outputs: + head_sha: ${{ steps.decide.outputs.head_sha }} + base_sha: ${{ steps.decide.outputs.base_sha }} + tools_sha: ${{ steps.decide.outputs.tools_sha }} + pr: ${{ steps.decide.outputs.pr }} + mode: ${{ steps.decide.outputs.mode }} + run: ${{ steps.decide.outputs.run }} + steps: + - id: decide + env: + GH_TOKEN: ${{ github.token }} + EVENT_LABEL: ${{ github.event.label.name }} + BASE_REF: ${{ github.event.pull_request.base.ref }} + run: | + set -e + # Resolve head/base entirely via the API (no checkout needed). + # PR events compare against the PR's base branch; dispatch + # compares against develop and can also validate A/A (base vs itself). + # + # tools_sha always names the commit that triggered this run, even + # in A/A mode where head_sha is overwritten to equal base_sha: the + # comparison script lives alongside the workflow file itself, and + # a merge-base predating this workflow's own introduction (as when + # A/A-validating the PR that adds benchmarking CI) may not have it. + if [ "${{ github.event_name }}" = "pull_request_target" ]; then + action="${{ github.event.action }}" + # Routed through env, not interpolated directly into the script, + # so an attacker-chosen label name can't break out of the string + # literal (only triage+ actors can name labels, but harden anyway). + event_label="$EVENT_LABEL" + # Unrelated label churn (e.g. "documentation") on a qualifying + # collaborator PR must not re-run benchmarks or, via + # cancel-in-progress, kill one already running. Checked before + # author association so it applies regardless of who the + # author is, and labels are left untouched here. + if [ "$action" = "labeled" ] && [ "$event_label" != "benchmark" ]; then + echo "run=false" >> "$GITHUB_OUTPUT" + echo "pr=${{ github.event.pull_request.number }}" >> "$GITHUB_OUTPUT" + echo "mode=ab" >> "$GITHUB_OUTPUT" + else + assoc="${{ github.event.pull_request.author_association }}" + label_present="${{ contains(github.event.pull_request.labels.*.name, 'benchmark') }}" + run=false + # COLLABORATOR = added to this repo; MEMBER (any cppalliance + # org member) is deliberately excluded — org membership is a + # wider circle than the people trusted with these machines. + # Members use the benchmark-label path like everyone else. + case "$assoc" in + OWNER|COLLABORATOR) run=true ;; + *) + # Test the label that fired THIS event, not "is benchmark + # present anywhere in the current label set" — the latter + # would let a stale label (e.g. a failed delete below) grant + # a run on ANY later labeled event, such as someone adding + # "needs-info", since the payload still lists "benchmark" + # among the PR's labels. Honoring only the labeled event + # itself also means a later synchronize can't ride an + # approval granted for an earlier diff. + if [ "$action" = "labeled" ] && [ "$event_label" = "benchmark" ]; then + run=true + fi + ;; + esac + # Strip the label whenever present, for any author — collaborators + # don't need it, and either way a lingering label is a stale- + # trigger reservoir. No `|| true`: under `set -e` a failed delete + # aborts here, before `run` is emitted, so bench's `run == 'true'` + # check sees an empty output and skips — a visible red gate job + # beats a silently stale label authorizing a future run. + if [ "$label_present" = "true" ]; then + gh api -X DELETE \ + "repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/labels/benchmark" + fi + echo "run=$run" >> "$GITHUB_OUTPUT" + head_sha="${{ github.event.pull_request.head.sha }}" + base_sha=$(gh api "repos/${{ github.repository }}/compare/${BASE_REF}...${head_sha}" \ + --jq .merge_base_commit.sha) + # pull_request_target runs workflow + job steps from the base + # branch, but env/expression context still carries PR data — + # tooling must come from the trusted base commit, never the + # untrusted PR head, or a malicious head could smuggle changes + # into compare.py/resolve-capy that this gate never reviewed. + tools_sha="${{ github.sha }}" + echo "pr=${{ github.event.pull_request.number }}" >> "$GITHUB_OUTPUT" + echo "mode=ab" >> "$GITHUB_OUTPUT" + fi + else + head_sha="${{ github.sha }}" + base_sha=$(gh api "repos/${{ github.repository }}/compare/develop...${head_sha}" \ + --jq .merge_base_commit.sha) + tools_sha="$head_sha" + if [ "${{ inputs.mode }}" = "aa" ]; then + head_sha="$base_sha" # A/A: both sides identical + fi + echo "run=true" >> "$GITHUB_OUTPUT" + echo "pr=" >> "$GITHUB_OUTPUT" + echo "mode=${{ inputs.mode }}" >> "$GITHUB_OUTPUT" + fi + echo "head_sha=$head_sha" >> "$GITHUB_OUTPUT" + echo "base_sha=$base_sha" >> "$GITHUB_OUTPUT" + echo "tools_sha=$tools_sha" >> "$GITHUB_OUTPUT" + + bench: + needs: gate + if: needs.gate.outputs.run == 'true' + strategy: + fail-fast: false + matrix: + include: + - platform: linux + labels: '["self-hosted", "Linux", "X64"]' + backends: "epoll uring" + - platform: windows + labels: '["self-hosted", "Windows", "X64"]' + backends: "iocp" + - platform: macos + labels: '["self-hosted", "macOS", "ARM64"]' + backends: "kqueue" + name: bench-${{ matrix.platform }} + runs-on: ${{ fromJSON(matrix.labels) }} + timeout-minutes: 150 + env: + BENCH_ITERATIONS: ${{ inputs.iterations || '5' }} + BENCH_DURATION: ${{ inputs.duration || '1.5' }} + BENCH_CATEGORY: ${{ inputs.category || '' }} + defaults: + run: + shell: bash + steps: + - name: Clean workspace + run: rm -rf results build-base build-head ws && mkdir -p results + + - name: Checkout head + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.head_sha }} + path: ws/head + persist-credentials: false + + - name: Checkout base + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.base_sha }} + path: ws/base + persist-credentials: false + + - name: Checkout tools (for compare.py) + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.tools_sha }} + path: ws/tools + persist-credentials: false + + - name: Resolve Capy branch + id: capy-ref + # Action definition must come from the trusted tools checkout, not + # the untrusted PR head, or a malicious head could rewrite this + # action to run arbitrary code under pull_request_target. + uses: ./ws/tools/.github/actions/resolve-capy + + - name: Checkout capy + uses: actions/checkout@v4 + with: + repository: ${{ steps.capy-ref.outputs.repo }} + ref: ${{ steps.capy-ref.outputs.ref }} + path: ws/capy + persist-credentials: false + + - name: Create capy provider + run: | + # Git Bash's `pwd` prints an MSYS path (e.g. /d/a/...) that the + # native cmake.exe can't resolve when embedded in a generated + # file; cygpath -m gives the drive-letter form cmake expects. + capy_dir="$(pwd)/ws/capy" + if command -v cygpath > /dev/null 2>&1; then + capy_dir="$(cygpath -m "$capy_dir")" + fi + cat > ws/provide-capy.cmake << EOF + set(BOOST_CAPY_BUILD_TESTS OFF CACHE BOOL "" FORCE) + set(BOOST_CAPY_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) + add_subdirectory("$capy_dir" + "\${CMAKE_BINARY_DIR}/_deps/capy-build" EXCLUDE_FROM_ALL) + EOF + + - name: Put common toolchain dirs on PATH + run: | + # Self-hosted runner service shells are often non-login/non-interactive + # and don't source the profile files an admin's interactive install + # left PATH changes in (.bashrc/.zprofile/etc). Widen discovery to the + # usual install locations without installing anything ourselves. + for dir in /opt/homebrew/bin /usr/local/bin "$HOME/.local/bin" /snap/bin /opt/cmake/bin; do + if [ -d "$dir" ]; then + echo "$dir" >> "$GITHUB_PATH" + fi + done + + - name: Toolchain diagnostics + run: | + echo "PATH=$PATH" + echo "-- command -v --" + command -v cmake || true + command -v git || true + command -v python3 || true + for dir in /opt/homebrew/bin /usr/local/bin "$HOME/.local/bin" /snap/bin /opt/cmake/bin; do + echo "-- ls $dir --" + ls -la "$dir" 2>/dev/null || echo "(missing)" + done + + - name: Build head and base + run: | + set -e + for side in head base; do + cmake -S "ws/$side" -B "build-$side" \ + -DCMAKE_BUILD_TYPE=Release \ + -DBOOST_COROSIO_BUILD_BENCH=ON \ + -DBOOST_COROSIO_BUILD_TESTS=OFF \ + -DBOOST_COROSIO_BUILD_EXAMPLES=OFF \ + -DCMAKE_PROJECT_boost_corosio_INCLUDE="$(pwd)/ws/provide-capy.cmake" + cmake --build "build-$side" --config Release \ + --target corosio_bench --parallel + done + + - name: Locate binaries + id: bins + run: | + set -e + for side in head base; do + bin=$(find "build-$side/bench" -type f \ + \( -name corosio_bench -o -name corosio_bench.exe \) | head -1) + test -n "$bin" + echo "$side=$bin" >> "$GITHUB_OUTPUT" + done + + - name: Run interleaved benchmarks + run: | + set -e + cat_flag="" + if [ -n "$BENCH_CATEGORY" ]; then + cat_flag="--category $BENCH_CATEGORY" + fi + for i in $(seq 1 "$BENCH_ITERATIONS"); do + # ABBA: odd iterations base first, even iterations head first + if [ $((i % 2)) -eq 1 ]; then order="base head"; else order="head base"; fi + for side in $order; do + if [ "$side" = base ]; then bin="${{ steps.bins.outputs.base }}"; + else bin="${{ steps.bins.outputs.head }}"; fi + for b in ${{ matrix.backends }}; do + "$bin" --library corosio --backend "$b" \ + --duration "$BENCH_DURATION" $cat_flag \ + --output "results/$side-$b-$i.json" + done + done + done + + - name: Summarize + run: | + python3 ws/tools/.github/bench/compare.py summarize \ + --platform "${{ matrix.platform }}" \ + --input-dir results \ + --output results/summary.json \ + --mode "${{ needs.gate.outputs.mode }}" + + - name: Upload results + uses: actions/upload-artifact@v4 + with: + name: bench-${{ matrix.platform }} + path: results/ + + report: + needs: [gate, bench] + # !cancelled(): still report when a bench job fails (partial results are + # useful), but never when the run is canceled — a superseded run must not + # overwrite the previous good comment with "no results". + if: ${{ !cancelled() && needs.gate.result == 'success' && needs.gate.outputs.run == 'true' }} + runs-on: ubuntu-24.04 + permissions: + contents: read + pull-requests: write + steps: + - name: Checkout tools (for compare.py) + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.tools_sha }} + persist-credentials: false + + - name: Download summaries + uses: actions/download-artifact@v4 + with: + pattern: bench-* + path: artifacts + + - name: Build report + run: | + python3 .github/bench/compare.py report \ + --summaries artifacts \ + --expect linux,windows,macos \ + --base-sha "${{ needs.gate.outputs.base_sha }}" \ + --head-sha "${{ needs.gate.outputs.head_sha }}" \ + --run-url "${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" \ + --output comment.md + + - name: Publish + env: + GH_TOKEN: ${{ github.token }} + PR: ${{ needs.gate.outputs.pr }} + run: | + set -e + if [ -z "$PR" ]; then + cat comment.md >> "$GITHUB_STEP_SUMMARY" + exit 0 + fi + marker="" + cid=$(gh api "repos/${{ github.repository }}/issues/$PR/comments?per_page=100" \ + --jq "[.[] | select(.user.login == \"github-actions[bot]\") + | select(.body | startswith(\"$marker\"))] | first | .id // empty") + if [ -n "$cid" ]; then + gh api -X PATCH "repos/${{ github.repository }}/issues/comments/$cid" \ + -F body=@comment.md + else + gh api "repos/${{ github.repository }}/issues/$PR/comments" \ + -F body=@comment.md + fi