diff --git a/.github/bench/compare.py b/.github/bench/compare.py new file mode 100644 index 000000000..bb526503d --- /dev/null +++ b/.github/bench/compare.py @@ -0,0 +1,383 @@ +#!/usr/bin/env python3 +"""Compare interleaved corosio benchmark runs. + +summarize: reduce raw per-iteration JSON (base/head × backend) to one +per-platform summary with flagging. +report: merge per-platform summaries into the PR comment markdown. + +Input files: --.json as written by +corosio_bench --output. Never exits nonzero because of suite-shape +differences between base and head; only real I/O or usage errors fail. + +Per benchmark, delta_pct is the median of the per-iteration paired +deltas (resists a single outlier iteration) and noise_pct is the +larger of the base and head sample CVs (a shift confined to one side +still sets a floor). +""" +import argparse +import json +import re +import statistics +import sys +from pathlib import Path + +MIN_EFFECT_PCT = 2.0 +NOISE_FACTOR = 3.0 +HIGHER_BETTER = ("bytes_per_sec", "items_per_sec", "ops_per_sec") +FNAME = re.compile(r"^(base|head)-([A-Za-z0-9_]+)-(\d+)\.json$") + + +def primary_metric(category, metrics): + """Pick the compared metric and its direction for one benchmark.""" + if "latency" in category and "latency_mean_ns" in metrics: + return "latency_mean_ns", "lower" + for m in HIGHER_BETTER: + if m in metrics: + return m, "higher" + if "latency_mean_ns" in metrics: + return "latency_mean_ns", "lower" + return None, None + + +def human(value, metric): + """Format a metric value with readable units.""" + if metric.endswith("_ns"): + for factor, unit in ((1e9, "s"), (1e6, "ms"), (1e3, "µs")): + if abs(value) >= factor: + return f"{value / factor:.2f} {unit}" + return f"{value:.0f} ns" + # Bytes glue the prefix to the unit (KB/s, GB/s); count-style metrics + # glue it to the number (1.82K ops/s) so sub-1000 values still carry a + # unit instead of a bare "/s". + if metric == "bytes_per_sec": + for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")): + if abs(value) >= factor: + n = value / factor + return (f"{n:.2f} {prefix}B/s" if n < 100 + else f"{n:.1f} {prefix}B/s") + return f"{value:.1f} B/s" + unit = "items/s" if metric == "items_per_sec" else "ops/s" + for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")): + if abs(value) >= factor: + n = value / factor + return (f"{n:.2f}{prefix} {unit}" if n < 100 + else f"{n:.1f}{prefix} {unit}") + return f"{value:.1f} {unit}" + + +MARKER = "" +MAX_COMMENT_CHARS = 60000 + + +FLAGGED_HEADER = "| Platform | Backend | Benchmark | Metric | Base | Head | Δ | Noise |" +FLAGGED_RULE = "|---|---|---|---|---|---|---|---|" +DETAIL_HEADER = "| Benchmark | Backend | Metric | Base | Head | Δ | Noise |" +DETAIL_RULE = "|---|---|---|---|---|---|---|" + + +def _delta_cells(r): + # Deltas are sign-normalized (positive = improvement), so color tracks + # verdict on every row, flagged or not. No green arrow exists in emoji, + # so a colored dot carries the verdict and a text arrowhead the direction. + # Color follows the two-decimal value actually shown, so a delta that + # displays as 0.00% is neutral rather than a red "-0.00%". + shown = round(r["delta_pct"], 2) + if shown == 0: + delta = "⚪ 0.00%" + else: + arrow = "🔴▼" if shown < 0 else "🟢▲" + delta = f"{arrow} {shown:+.2f}%" + noise = "—" if r["noise_pct"] is None else f"{r['noise_pct']:.2f}%" + return (f"{human(r['base_mean'], r['metric'])} " + f"| {human(r['head_mean'], r['metric'])} " + f"| {delta} | {noise}") + + +def _flagged_line(platform, r): + return (f"| {platform} | {r['backend']} | {r['category']}/{r['name']} " + f"| {r['metric']} | {_delta_cells(r)} |") + + +def _detail_line(r): + return f"| {r['name']} | {r['backend']} | {r['metric']} | {_delta_cells(r)} |" + + +def _platform_detail_lines(platform, s, condensed): + """Full per-platform results table, or a one-line summary when condensed.""" + if condensed: + counts = [] + if s["new"]: + counts.append(f"{len(s['new'])} new") + if s["removed"]: + counts.append(f"{len(s['removed'])} removed") + if s["unsupported"]: + counts.append(f"{len(s['unsupported'])} unsupported") + suffix = f" ({', '.join(counts)})" if counts else "" + return [f"- **{platform}** — {len(s['rows'])} benchmarks, " + f"{s['iterations']} iterations, {s['duration_s']}s each" + f"{suffix}"] + + lines = [f"
{platform} — full results " + f"({len(s['rows'])} benchmarks, {s['iterations']} iterations, " + f"{s['duration_s']}s each)", ""] + # One table per category keeps rows narrow enough for GitHub's comment + # width; same-name rows sort adjacently so backends compare at a glance. + by_category = {} + for r in s["rows"]: + by_category.setdefault(r["category"], []).append(r) + for category in sorted(by_category): + lines += [f"**{category}**", "", DETAIL_HEADER, DETAIL_RULE] + rows = sorted(by_category[category], + key=lambda r: (r["name"], r["backend"])) + lines += [_detail_line(r) for r in rows] + lines.append("") + if s["new"]: + lines += ["**New benchmarks (no baseline):**", ""] + lines += [f"- `{n['category']}/{n['name']}` [{n['backend']}] " + f"{human(n['head_mean'], n['metric'])}" for n in s["new"]] + lines.append("") + if s["removed"]: + lines += ["**Removed benchmarks:** " + + ", ".join(f"`{r['category']}/{r['name']}`" + for r in s["removed"]), ""] + if s["unsupported"]: + lines += ["**Unsupported (no recognized metric):** " + + ", ".join(f"`{u['category']}/{u['name']}`" + for u in s["unsupported"]), ""] + lines += ["
", ""] + return lines + + +def _build_report(summaries, base_sha, head_sha, run_url, condensed): + lines = [MARKER, "## Benchmark report", ""] + mode = next((s["mode"] for s in summaries.values() if s), "ab") + if mode == "aa": + lines += ["**A/A validation run** — base compared against itself; " + "every flag below is a false positive.", ""] + lines += [f"`{base_sha[:12]}` (base) vs `{head_sha[:12]}` (head)", ""] + + for platform, s in summaries.items(): + if s is None: + lines.append(f"- ❌ **{platform}** — no results " + "(job failed or runner offline)") + elif s["flagged_count"]: + lines.append(f"- ⚠️ **{platform}** — {s['flagged_count']} flagged " + f"({', '.join(s['backends'])})") + else: + lines.append(f"- ✅ **{platform}** — clean " + f"({', '.join(s['backends'])})") + lines.append("") + + flagged = [(p, r) for p, s in summaries.items() if s + for r in s["rows"] if r["flagged"]] + if flagged: + lines += ["### ⚠️ Flagged", "", FLAGGED_HEADER, FLAGGED_RULE] + lines += [_flagged_line(p, r) for p, r in flagged] + lines.append("") + + for platform, s in summaries.items(): + if s is None: + continue + lines += _platform_detail_lines(platform, s, condensed) + + if condensed: + lines += ["_full tables omitted — comment size limit; " + "see the run artifacts_", ""] + + lines += [f"[Run & raw JSON artifacts]({run_url}) · " + "flag rule: |median Δ| > max(2%, 3×CV of the noisier side) " + "· advisory only"] + return "\n".join(lines) + "\n" + + +def report(summaries, base_sha, head_sha, run_url): + md = _build_report(summaries, base_sha, head_sha, run_url, condensed=False) + # GitHub caps issue comments at 65536 chars; a run with many benchmarks + # across three platforms can exceed that in full-details form. + if len(md) > MAX_COMMENT_CHARS: + md = _build_report(summaries, base_sha, head_sha, run_url, condensed=True) + return md + + +def load_runs(input_dir): + """Return {(side, backend, iter): {(category, name): {metric: value}}}.""" + runs = {} + for p in sorted(Path(input_dir).iterdir()): + m = FNAME.match(p.name) + if not m: + continue + side, backend, it = m.group(1), m.group(2), int(m.group(3)) + try: + payload = json.loads(p.read_text()) + except (OSError, json.JSONDecodeError) as e: + print(f"warning: skipping unreadable {p.name}: {e}", file=sys.stderr) + continue + if not isinstance(payload, dict): + print(f"warning: skipping non-object JSON {p.name}", file=sys.stderr) + continue + benchmarks = payload.get("benchmarks", []) + if not isinstance(benchmarks, list): + print(f"warning: skipping {p.name}: benchmarks field is not a list", file=sys.stderr) + continue + table = {} + for b in benchmarks: + if not isinstance(b, dict): + print(f"warning: skipping non-object benchmark entry in {p.name}", file=sys.stderr) + continue + key = (b.get("category", ""), b.get("name", "")) + table[key] = { + k: v for k, v in b.items() + if isinstance(v, (int, float)) and not isinstance(v, bool) + } + runs[(side, backend, it)] = table + return runs + + +def _values(runs, side, backend, key, metric): + """Metric samples for one benchmark on one side, ordered by iteration.""" + out = [] + for (s, b, it), table in sorted(runs.items(), key=lambda kv: kv[0][2]): + if s == side and b == backend and key in table and metric in table[key]: + out.append((it, table[key][metric])) + return out + + +def summarize(input_dir, platform, mode="ab"): + runs = load_runs(input_dir) + backends = sorted({b for (_, b, _) in runs}) + iterations = max((it for (_, _, it) in runs), default=0) + duration = 0.0 + rows, new, removed, unsupported = [], [], [], [] + + for backend in backends: + base_keys, head_keys = set(), set() + sample = {} + for (s, b, it), table in runs.items(): + if b != backend: + continue + (base_keys if s == "base" else head_keys).update(table) + for key, metrics in table.items(): + sample.setdefault(key, metrics) + + for key in sorted(base_keys | head_keys): + category, name = key + metric, direction = primary_metric(category, sample.get(key, {})) + if metric is None: + unsupported.append( + {"backend": backend, "category": category, "name": name}) + continue + if key not in base_keys: + head = _values(runs, "head", backend, key, metric) + mean = statistics.fmean(v for _, v in head) if head else 0.0 + new.append({"backend": backend, "category": category, + "name": name, "metric": metric, "head_mean": mean}) + continue + if key not in head_keys: + removed.append( + {"backend": backend, "category": category, "name": name}) + continue + + base = dict(_values(runs, "base", backend, key, metric)) + head = dict(_values(runs, "head", backend, key, metric)) + common = sorted(set(base) & set(head)) + deltas = [] + for it in common: + b_v, h_v = base[it], head[it] + if b_v == 0: + continue + d = (h_v - b_v) / b_v * 100.0 + if direction == "lower": + d = -d + deltas.append(d) + if not deltas: + unsupported.append( + {"backend": backend, "category": category, "name": name}) + continue + + base_vals = [base[it] for it in common] + head_vals = [head[it] for it in common] + base_mean = statistics.fmean(base_vals) + head_mean = statistics.fmean(head_vals) + + def cv(vals, mean): + if len(vals) >= 2 and mean != 0: + return statistics.stdev(vals) / abs(mean) * 100.0 + return None + + base_cv = cv(base_vals, base_mean) + head_cv = cv(head_vals, head_mean) + # a single-side outlier or a side-level shift can leave one + # side's spread tight while the other carries the noise + noise_candidates = [c for c in (base_cv, head_cv) if c is not None] + noise_pct = max(noise_candidates) if noise_candidates else None + # median resists a single blown-up iteration that a mean would not + delta_pct = statistics.median(deltas) + flagged = ( + noise_pct is not None + and abs(delta_pct) > max(MIN_EFFECT_PCT, NOISE_FACTOR * noise_pct) + ) + rows.append({ + "backend": backend, "category": category, "name": name, + "metric": metric, "direction": direction, + "base_mean": base_mean, "head_mean": head_mean, + "delta_pct": delta_pct, "noise_pct": noise_pct, + "flagged": flagged, + }) + + for p in Path(input_dir).iterdir(): + if FNAME.match(p.name): + try: + duration = json.loads(p.read_text())["metadata"]["duration_s"] + break + except Exception: + pass + + return { + "platform": platform, "backends": backends, + "iterations": iterations, "duration_s": duration, "mode": mode, + "rows": rows, "new": new, "removed": removed, + "unsupported": unsupported, + "flagged_count": sum(1 for r in rows if r["flagged"]), + } + + +def main(argv=None): + ap = argparse.ArgumentParser(prog="compare.py") + sub = ap.add_subparsers(dest="cmd", required=True) + s = sub.add_parser("summarize") + s.add_argument("--platform", required=True) + s.add_argument("--input-dir", required=True) + s.add_argument("--output", required=True) + s.add_argument("--mode", default="ab", choices=("ab", "aa")) + r = sub.add_parser("report") + r.add_argument("--summaries", required=True, + help="dir containing bench-/summary.json") + r.add_argument("--expect", required=True, + help="comma-separated platform list") + r.add_argument("--base-sha", required=True) + r.add_argument("--head-sha", required=True) + r.add_argument("--run-url", required=True) + r.add_argument("--output", required=True) + args = ap.parse_args(argv) + + if args.cmd == "summarize": + summary = summarize(args.input_dir, args.platform, args.mode) + Path(args.output).write_text(json.dumps(summary, indent=2)) + print(f"{args.platform}: {len(summary['rows'])} rows, " + f"{summary['flagged_count']} flagged") + elif args.cmd == "report": + summaries = {} + for platform in args.expect.split(","): + p = Path(args.summaries) / f"bench-{platform}" / "summary.json" + try: + summaries[platform] = json.loads(p.read_text()) + except (OSError, json.JSONDecodeError): + summaries[platform] = None + md = report(summaries, args.base_sha, args.head_sha, args.run_url) + Path(args.output).write_text(md) + print(f"report written: {args.output}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml new file mode 100644 index 000000000..51e87080a --- /dev/null +++ b/.github/workflows/benchmarks.yml @@ -0,0 +1,381 @@ +# +# Copyright (c) 2026 Steve Gerbino +# +# Distributed under the Boost Software License, Version 1.0. (See accompanying +# file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt) +# +# Official repository: https://github.com/cppalliance/corosio/ +# +# Per-PR performance benchmarks on dedicated self-hosted runners. +# Advisory only. See issue #343. +# +# Runner prerequisites (per machine, maintained by the infra admin): +# all : git, cmake >= 3.20, ninja or make/msbuild, python3 +# linux : gcc or clang, liburing-dev +# windows : MSVC (vcvars auto-detected by cmake), python3 on PATH, +# Git Bash (for shell: bash steps) +# macos : Xcode command line tools + +name: benchmarks + +on: + pull_request_target: + types: [opened, reopened, synchronize, labeled] + paths: + - 'include/**' + - 'src/**' + - 'bench/**' + - '.github/workflows/benchmarks.yml' + - '.github/bench/**' + workflow_dispatch: + inputs: + mode: + description: "ab = head vs merge-base, aa = base vs itself" + type: choice + options: [ab, aa] + default: ab + iterations: + description: "override BENCH_ITERATIONS" + default: "" + duration: + description: "override BENCH_DURATION" + default: "" + category: + description: "run only this benchmark category (default: all)" + default: "" + +concurrency: + group: benchmarks-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + gate: + runs-on: ubuntu-24.04 + permissions: + contents: read + pull-requests: write + outputs: + head_sha: ${{ steps.decide.outputs.head_sha }} + base_sha: ${{ steps.decide.outputs.base_sha }} + tools_sha: ${{ steps.decide.outputs.tools_sha }} + pr: ${{ steps.decide.outputs.pr }} + mode: ${{ steps.decide.outputs.mode }} + run: ${{ steps.decide.outputs.run }} + steps: + - id: decide + env: + GH_TOKEN: ${{ github.token }} + EVENT_LABEL: ${{ github.event.label.name }} + BASE_REF: ${{ github.event.pull_request.base.ref }} + run: | + set -e + # Resolve head/base entirely via the API (no checkout needed). + # PR events compare against the PR's base branch; dispatch + # compares against develop and can also validate A/A (base vs itself). + # + # tools_sha always names the commit that triggered this run, even + # in A/A mode where head_sha is overwritten to equal base_sha: the + # comparison script lives alongside the workflow file itself, and + # a merge-base predating this workflow's own introduction (as when + # A/A-validating the PR that adds benchmarking CI) may not have it. + if [ "${{ github.event_name }}" = "pull_request_target" ]; then + action="${{ github.event.action }}" + # Routed through env, not interpolated directly into the script, + # so an attacker-chosen label name can't break out of the string + # literal (only triage+ actors can name labels, but harden anyway). + event_label="$EVENT_LABEL" + # Unrelated label churn (e.g. "documentation") on a qualifying + # collaborator PR must not re-run benchmarks or, via + # cancel-in-progress, kill one already running. Checked before + # author association so it applies regardless of who the + # author is, and labels are left untouched here. + if [ "$action" = "labeled" ] && [ "$event_label" != "benchmark" ]; then + echo "run=false" >> "$GITHUB_OUTPUT" + echo "pr=${{ github.event.pull_request.number }}" >> "$GITHUB_OUTPUT" + echo "mode=ab" >> "$GITHUB_OUTPUT" + else + assoc="${{ github.event.pull_request.author_association }}" + label_present="${{ contains(github.event.pull_request.labels.*.name, 'benchmark') }}" + run=false + # COLLABORATOR = added to this repo; MEMBER (any cppalliance + # org member) is deliberately excluded — org membership is a + # wider circle than the people trusted with these machines. + # Members use the benchmark-label path like everyone else. + case "$assoc" in + OWNER|COLLABORATOR) run=true ;; + *) + # Test the label that fired THIS event, not "is benchmark + # present anywhere in the current label set" — the latter + # would let a stale label (e.g. a failed delete below) grant + # a run on ANY later labeled event, such as someone adding + # "needs-info", since the payload still lists "benchmark" + # among the PR's labels. Honoring only the labeled event + # itself also means a later synchronize can't ride an + # approval granted for an earlier diff. + if [ "$action" = "labeled" ] && [ "$event_label" = "benchmark" ]; then + run=true + fi + ;; + esac + # Strip the label whenever present, for any author — collaborators + # don't need it, and either way a lingering label is a stale- + # trigger reservoir. No `|| true`: under `set -e` a failed delete + # aborts here, before `run` is emitted, so bench's `run == 'true'` + # check sees an empty output and skips — a visible red gate job + # beats a silently stale label authorizing a future run. + if [ "$label_present" = "true" ]; then + gh api -X DELETE \ + "repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/labels/benchmark" + fi + echo "run=$run" >> "$GITHUB_OUTPUT" + head_sha="${{ github.event.pull_request.head.sha }}" + base_sha=$(gh api "repos/${{ github.repository }}/compare/${BASE_REF}...${head_sha}" \ + --jq .merge_base_commit.sha) + # pull_request_target runs workflow + job steps from the base + # branch, but env/expression context still carries PR data — + # tooling must come from the trusted base commit, never the + # untrusted PR head, or a malicious head could smuggle changes + # into compare.py/resolve-capy that this gate never reviewed. + tools_sha="${{ github.sha }}" + echo "pr=${{ github.event.pull_request.number }}" >> "$GITHUB_OUTPUT" + echo "mode=ab" >> "$GITHUB_OUTPUT" + fi + else + head_sha="${{ github.sha }}" + base_sha=$(gh api "repos/${{ github.repository }}/compare/develop...${head_sha}" \ + --jq .merge_base_commit.sha) + tools_sha="$head_sha" + if [ "${{ inputs.mode }}" = "aa" ]; then + head_sha="$base_sha" # A/A: both sides identical + fi + echo "run=true" >> "$GITHUB_OUTPUT" + echo "pr=" >> "$GITHUB_OUTPUT" + echo "mode=${{ inputs.mode }}" >> "$GITHUB_OUTPUT" + fi + echo "head_sha=$head_sha" >> "$GITHUB_OUTPUT" + echo "base_sha=$base_sha" >> "$GITHUB_OUTPUT" + echo "tools_sha=$tools_sha" >> "$GITHUB_OUTPUT" + + bench: + needs: gate + if: needs.gate.outputs.run == 'true' + strategy: + fail-fast: false + matrix: + include: + - platform: linux + labels: '["self-hosted", "Linux", "X64"]' + backends: "epoll uring" + - platform: windows + labels: '["self-hosted", "Windows", "X64"]' + backends: "iocp" + - platform: macos + labels: '["self-hosted", "macOS", "ARM64"]' + backends: "kqueue" + name: bench-${{ matrix.platform }} + runs-on: ${{ fromJSON(matrix.labels) }} + timeout-minutes: 150 + env: + BENCH_ITERATIONS: ${{ inputs.iterations || '5' }} + BENCH_DURATION: ${{ inputs.duration || '1.5' }} + BENCH_CATEGORY: ${{ inputs.category || '' }} + defaults: + run: + shell: bash + steps: + - name: Clean workspace + run: rm -rf results build-base build-head ws && mkdir -p results + + - name: Checkout head + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.head_sha }} + path: ws/head + persist-credentials: false + + - name: Checkout base + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.base_sha }} + path: ws/base + persist-credentials: false + + - name: Checkout tools (for compare.py) + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.tools_sha }} + path: ws/tools + persist-credentials: false + + - name: Resolve Capy branch + id: capy-ref + # Action definition must come from the trusted tools checkout, not + # the untrusted PR head, or a malicious head could rewrite this + # action to run arbitrary code under pull_request_target. + uses: ./ws/tools/.github/actions/resolve-capy + + - name: Checkout capy + uses: actions/checkout@v4 + with: + repository: ${{ steps.capy-ref.outputs.repo }} + ref: ${{ steps.capy-ref.outputs.ref }} + path: ws/capy + persist-credentials: false + + - name: Create capy provider + run: | + # Git Bash's `pwd` prints an MSYS path (e.g. /d/a/...) that the + # native cmake.exe can't resolve when embedded in a generated + # file; cygpath -m gives the drive-letter form cmake expects. + capy_dir="$(pwd)/ws/capy" + if command -v cygpath > /dev/null 2>&1; then + capy_dir="$(cygpath -m "$capy_dir")" + fi + cat > ws/provide-capy.cmake << EOF + set(BOOST_CAPY_BUILD_TESTS OFF CACHE BOOL "" FORCE) + set(BOOST_CAPY_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) + add_subdirectory("$capy_dir" + "\${CMAKE_BINARY_DIR}/_deps/capy-build" EXCLUDE_FROM_ALL) + EOF + + - name: Put common toolchain dirs on PATH + run: | + # Self-hosted runner service shells are often non-login/non-interactive + # and don't source the profile files an admin's interactive install + # left PATH changes in (.bashrc/.zprofile/etc). Widen discovery to the + # usual install locations without installing anything ourselves. + for dir in /opt/homebrew/bin /usr/local/bin "$HOME/.local/bin" /snap/bin /opt/cmake/bin; do + if [ -d "$dir" ]; then + echo "$dir" >> "$GITHUB_PATH" + fi + done + + - name: Toolchain diagnostics + run: | + echo "PATH=$PATH" + echo "-- command -v --" + command -v cmake || true + command -v git || true + command -v python3 || true + for dir in /opt/homebrew/bin /usr/local/bin "$HOME/.local/bin" /snap/bin /opt/cmake/bin; do + echo "-- ls $dir --" + ls -la "$dir" 2>/dev/null || echo "(missing)" + done + + - name: Build head and base + run: | + set -e + for side in head base; do + cmake -S "ws/$side" -B "build-$side" \ + -DCMAKE_BUILD_TYPE=Release \ + -DBOOST_COROSIO_BUILD_BENCH=ON \ + -DBOOST_COROSIO_BUILD_TESTS=OFF \ + -DBOOST_COROSIO_BUILD_EXAMPLES=OFF \ + -DCMAKE_PROJECT_boost_corosio_INCLUDE="$(pwd)/ws/provide-capy.cmake" + cmake --build "build-$side" --config Release \ + --target corosio_bench --parallel + done + + - name: Locate binaries + id: bins + run: | + set -e + for side in head base; do + bin=$(find "build-$side/bench" -type f \ + \( -name corosio_bench -o -name corosio_bench.exe \) | head -1) + test -n "$bin" + echo "$side=$bin" >> "$GITHUB_OUTPUT" + done + + - name: Run interleaved benchmarks + run: | + set -e + cat_flag="" + if [ -n "$BENCH_CATEGORY" ]; then + cat_flag="--category $BENCH_CATEGORY" + fi + for i in $(seq 1 "$BENCH_ITERATIONS"); do + # ABBA: odd iterations base first, even iterations head first + if [ $((i % 2)) -eq 1 ]; then order="base head"; else order="head base"; fi + for side in $order; do + if [ "$side" = base ]; then bin="${{ steps.bins.outputs.base }}"; + else bin="${{ steps.bins.outputs.head }}"; fi + for b in ${{ matrix.backends }}; do + "$bin" --library corosio --backend "$b" \ + --duration "$BENCH_DURATION" $cat_flag \ + --output "results/$side-$b-$i.json" + done + done + done + + - name: Summarize + run: | + python3 ws/tools/.github/bench/compare.py summarize \ + --platform "${{ matrix.platform }}" \ + --input-dir results \ + --output results/summary.json \ + --mode "${{ needs.gate.outputs.mode }}" + + - name: Upload results + uses: actions/upload-artifact@v4 + with: + name: bench-${{ matrix.platform }} + path: results/ + + report: + needs: [gate, bench] + # !cancelled(): still report when a bench job fails (partial results are + # useful), but never when the run is canceled — a superseded run must not + # overwrite the previous good comment with "no results". + if: ${{ !cancelled() && needs.gate.result == 'success' && needs.gate.outputs.run == 'true' }} + runs-on: ubuntu-24.04 + permissions: + contents: read + pull-requests: write + steps: + - name: Checkout tools (for compare.py) + uses: actions/checkout@v4 + with: + ref: ${{ needs.gate.outputs.tools_sha }} + persist-credentials: false + + - name: Download summaries + uses: actions/download-artifact@v4 + with: + pattern: bench-* + path: artifacts + + - name: Build report + run: | + python3 .github/bench/compare.py report \ + --summaries artifacts \ + --expect linux,windows,macos \ + --base-sha "${{ needs.gate.outputs.base_sha }}" \ + --head-sha "${{ needs.gate.outputs.head_sha }}" \ + --run-url "${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" \ + --output comment.md + + - name: Publish + env: + GH_TOKEN: ${{ github.token }} + PR: ${{ needs.gate.outputs.pr }} + run: | + set -e + if [ -z "$PR" ]; then + cat comment.md >> "$GITHUB_STEP_SUMMARY" + exit 0 + fi + marker="" + cid=$(gh api "repos/${{ github.repository }}/issues/$PR/comments?per_page=100" \ + --jq "[.[] | select(.user.login == \"github-actions[bot]\") + | select(.body | startswith(\"$marker\"))] | first | .id // empty") + if [ -n "$cid" ]; then + gh api -X PATCH "repos/${{ github.repository }}/issues/comments/$cid" \ + -F body=@comment.md + else + gh api "repos/${{ github.repository }}/issues/$PR/comments" \ + -F body=@comment.md + fi