From a1a2ba8276de0906d4940d129a30e03652edc75e Mon Sep 17 00:00:00 2001 From: Steve Gerbino Date: Thu, 1 Oct 2026 21:59:52 +0200 Subject: [PATCH 1/3] bench: skip local socket suites on Windows local::connect_pair relies on socketpair(), which has no Windows implementation and throws operation_not_supported at runtime, taking the whole comparison run down with it. The corosio local socket suites are already POSIX-only, so Windows loses no comparison coverage. --- bench/main.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/bench/main.cpp b/bench/main.cpp index c6e50ca6d..9cb249a30 100644 --- a/bench/main.cpp +++ b/bench/main.cpp @@ -104,8 +104,11 @@ add_asio_suites(bench::benchmark_runner& runner) runner.add_suite("asio", asio_bench::make_http_server_suite()); runner.add_suite("asio", asio_bench::make_accept_churn_suite()); runner.add_suite("asio", asio_bench::make_fan_out_suite()); + // connect_pair needs socketpair(), which Windows lacks +#if BOOST_COROSIO_POSIX runner.add_suite("asio", asio_bench::make_local_socket_throughput_suite()); runner.add_suite("asio", asio_bench::make_local_socket_latency_suite()); +#endif } void @@ -123,12 +126,14 @@ add_asio_callback_suites(bench::benchmark_runner& runner) "asio_callback", asio_callback_bench::make_accept_churn_suite()); runner.add_suite( "asio_callback", asio_callback_bench::make_fan_out_suite()); +#if BOOST_COROSIO_POSIX runner.add_suite( "asio_callback", asio_callback_bench::make_local_socket_throughput_suite()); runner.add_suite( "asio_callback", asio_callback_bench::make_local_socket_latency_suite()); +#endif } #endif From f9f3f29b54a027c26c861ab708ec7d005597a406 Mon Sep 17 00:00:00 2001 From: Steve Gerbino Date: Fri, 2 Oct 2026 15:48:48 +0200 Subject: [PATCH 2/3] bench: prevent TIME_WAIT pileup in the churn benchmarks Only the connecting side carried linger(0), so on loopback a server FIN racing the client RST could still park connections in TIME_WAIT. At production durations the leak overflowed the macOS PCB table mid-benchmark (70k against a 12k drain threshold), socket creation started failing, and the corosio loop ignored the open() error before set_option threw and took down the whole suite. The listener now carries the churn socket options -- accepted sockets inherit them on every supported platform -- so both sides RST at close without adding per-connection syscalls to the measured loop, and a failed open() now ends the benchmark instead of aborting the process. --- bench/asio/callback/accept_churn_bench.cpp | 8 ++++++ bench/asio/coroutine/accept_churn_bench.cpp | 8 ++++++ bench/corosio/accept_churn_bench.cpp | 31 +++++++++++++++++---- 3 files changed, 42 insertions(+), 5 deletions(-) diff --git a/bench/asio/callback/accept_churn_bench.cpp b/bench/asio/callback/accept_churn_bench.cpp index 9d68c7110..32d1ddb71 100644 --- a/bench/asio/callback/accept_churn_bench.cpp +++ b/bench/asio/callback/accept_churn_bench.cpp @@ -60,6 +60,14 @@ make_churn_acceptor(asio::io_context& ioc) ec = acc.open(tcp::v4(), ec); if (!ec) ec = acc.set_option(tcp_acceptor::reuse_address(true), ec); + // Accepted sockets inherit these from the listener + if (!ec) + ec = acc.set_option(asio::socket_base::send_buffer_size(1024), ec); + if (!ec) + ec = acc.set_option( + asio::socket_base::receive_buffer_size(1024), ec); + if (!ec) + ec = acc.set_option(asio::socket_base::linger(true, 0), ec); if (!ec) ec = acc.bind(tcp::endpoint(tcp::v4(), 0), ec); if (!ec) diff --git a/bench/asio/coroutine/accept_churn_bench.cpp b/bench/asio/coroutine/accept_churn_bench.cpp index 7b70f225d..e24745205 100644 --- a/bench/asio/coroutine/accept_churn_bench.cpp +++ b/bench/asio/coroutine/accept_churn_bench.cpp @@ -52,6 +52,14 @@ make_churn_acceptor(asio::io_context& ioc) ec = acc.open(tcp::v4(), ec); if (!ec) ec = acc.set_option(tcp_acceptor::reuse_address(true), ec); + // Accepted sockets inherit these from the listener + if (!ec) + ec = acc.set_option(asio::socket_base::send_buffer_size(1024), ec); + if (!ec) + ec = acc.set_option( + asio::socket_base::receive_buffer_size(1024), ec); + if (!ec) + ec = acc.set_option(asio::socket_base::linger(true, 0), ec); if (!ec) ec = acc.bind(tcp::endpoint(tcp::v4(), 0), ec); if (!ec) diff --git a/bench/corosio/accept_churn_bench.cpp b/bench/corosio/accept_churn_bench.cpp index 434100717..858d94b63 100644 --- a/bench/corosio/accept_churn_bench.cpp +++ b/bench/corosio/accept_churn_bench.cpp @@ -43,6 +43,17 @@ configure_churn_socket(corosio::tcp_socket& s) s.set_option(corosio::native_socket_option::linger(true, 0)); } +// Accepted sockets inherit buffer sizes and linger from the listener, +// so the server side of each churn connection needs no per-accept setup +template +static void +configure_churn_acceptor(Acceptor& acc) +{ + acc.set_option(corosio::native_socket_option::send_buffer_size(1024)); + acc.set_option(corosio::native_socket_option::receive_buffer_size(1024)); + acc.set_option(corosio::native_socket_option::linger(true, 0)); +} + // Single connect/accept/1-byte-exchange/close loop template void @@ -55,6 +66,7 @@ bench_sequential_churn(bench::state& state) acceptor_type acc(ioc); std::ignore = acc.open(); acc.set_option(corosio::native_socket_option::reuse_address(true)); + configure_churn_acceptor(acc); if (auto ec = acc.bind(corosio::endpoint(corosio::ipv4_address::loopback(), 0))) @@ -77,7 +89,8 @@ bench_sequential_churn(bench::state& state) socket_type client(ioc); socket_type server(ioc); - std::ignore = client.open(); + if (client.open()) + co_return; configure_churn_socket(client); capy::run_async(ioc.get_executor())( @@ -137,6 +150,7 @@ bench_sequential_churn_lockless(bench::state& state) acceptor_type acc(ioc); std::ignore = acc.open(); acc.set_option(corosio::native_socket_option::reuse_address(true)); + configure_churn_acceptor(acc); if (auto ec = acc.bind(corosio::endpoint(corosio::ipv4_address::loopback(), 0))) @@ -159,7 +173,8 @@ bench_sequential_churn_lockless(bench::state& state) socket_type client(ioc); socket_type server(ioc); - std::ignore = client.open(); + if (client.open()) + co_return; configure_churn_socket(client); capy::run_async(ioc.get_executor())( @@ -227,6 +242,7 @@ bench_concurrent_churn(bench::state& state) auto& acc = acceptors.back(); std::ignore = acc.open(); acc.set_option(corosio::native_socket_option::reuse_address(true)); + configure_churn_acceptor(acc); if (auto ec = acc.bind( corosio::endpoint(corosio::ipv4_address::loopback(), 0))) { @@ -250,7 +266,8 @@ bench_concurrent_churn(bench::state& state) socket_type client(ioc); socket_type server(ioc); - std::ignore = client.open(); + if (client.open()) + co_return; configure_churn_socket(client); capy::run_async(ioc.get_executor())( @@ -314,6 +331,7 @@ bench_burst_churn(bench::state& state) acceptor_type acc(ioc); std::ignore = acc.open(); acc.set_option(corosio::native_socket_option::reuse_address(true)); + configure_churn_acceptor(acc); if (auto ec = acc.bind(corosio::endpoint(corosio::ipv4_address::loopback(), 0))) @@ -342,7 +360,8 @@ bench_burst_churn(bench::state& state) for (int i = 0; i < burst_size; ++i) { clients.emplace_back(ioc); - std::ignore = clients.back().open(); + if (clients.back().open()) + co_return; configure_churn_socket(clients.back()); capy::run_async(ioc.get_executor())( [](socket_type& c, corosio::endpoint ep) -> capy::task<> { @@ -399,6 +418,7 @@ bench_burst_churn_lockless(bench::state& state) acceptor_type acc(ioc); std::ignore = acc.open(); acc.set_option(corosio::native_socket_option::reuse_address(true)); + configure_churn_acceptor(acc); if (auto ec = acc.bind(corosio::endpoint(corosio::ipv4_address::loopback(), 0))) @@ -427,7 +447,8 @@ bench_burst_churn_lockless(bench::state& state) for (int i = 0; i < burst_size; ++i) { clients.emplace_back(ioc); - std::ignore = clients.back().open(); + if (clients.back().open()) + co_return; configure_churn_socket(clients.back()); capy::run_async(ioc.get_executor())( [](socket_type& c, corosio::endpoint ep) -> capy::task<> { From 99008d23794fe4fc353c66f18aa79733f867504d Mon Sep 17 00:00:00 2001 From: Steve Gerbino Date: Fri, 2 Oct 2026 05:31:33 +0200 Subject: [PATCH 3/3] bench: publish a generated, reproducible benchmark report Replaces the hand-written February benchmark page with a pipeline that regenerates it from fresh runs on the three dedicated machines (closes #352): - benchmark-report.yml: dispatch-only workflow running the comparison suite (corosio, asio coroutines, asio callbacks) on every platform, uploading raw per-iteration JSON and an environment capture. Linux builds asio against both epoll and io_uring so each corosio configuration is measured against a same-reactor asio. - report_page.py: stdlib-only generator producing the four Antora pages and theme-aware inline SVG charts; medians across iterations, CV-based noise classification, diverging bars against the asio-coroutines baseline. - Scenario descriptions in the bench harness (describe and category_description), rendered as category intros and chart hover text. - bench/README.md: how to run the suite, dispatch the workflow, and regenerate the pages. - Generated pages and charts from a full 7x2 production run; the old benchmark-report.adoc URL survives via a page alias, nav gains the per-platform entries, and the lint baseline follows the page swap. --- .github/bench/report_page.py | 1375 +++++++++++++++++ .github/workflows/benchmark-report.yml | 229 +++ bench/README.md | 150 ++ bench/common/benchmark.hpp | 29 + bench/common/suite.hpp | 36 +- bench/corosio/accept_churn_bench.cpp | 20 + bench/corosio/fan_out_bench.cpp | 23 + bench/corosio/http_server_bench.cpp | 16 + bench/corosio/io_context_bench.cpp | 32 +- bench/corosio/local_socket_latency_bench.cpp | 16 + .../corosio/local_socket_throughput_bench.cpp | 15 + bench/corosio/socket_latency_bench.cpp | 16 + bench/corosio/socket_throughput_bench.cpp | 20 + .../config/vocabularies/Corosio/accept.txt | 3 + doc/lint/baseline.json | 12 +- .../images/bench/linux-accept_churn-epoll.svg | 51 + .../images/bench/linux-accept_churn-uring.svg | 51 + .../images/bench/linux-fan_out-epoll-g1.svg | 50 + .../images/bench/linux-fan_out-epoll-g2.svg | 60 + .../images/bench/linux-fan_out-uring-g1.svg | 52 + .../images/bench/linux-fan_out-uring-g2.svg | 58 + .../images/bench/linux-http_server-epoll.svg | 59 + .../images/bench/linux-http_server-uring.svg | 57 + .../linux-local_socket_latency-epoll.svg | 62 + .../linux-local_socket_latency-uring.svg | 58 + ...linux-local_socket_throughput-epoll-g1.svg | 62 + ...linux-local_socket_throughput-epoll-g2.svg | 58 + ...linux-local_socket_throughput-uring-g1.svg | 58 + ...linux-local_socket_throughput-uring-g2.svg | 60 + .../bench/linux-socket_latency-epoll.svg | 58 + .../bench/linux-socket_latency-uring.svg | 58 + .../linux-socket_throughput-epoll-g1.svg | 58 + .../linux-socket_throughput-epoll-g2.svg | 35 + .../linux-socket_throughput-epoll-g3.svg | 60 + .../linux-socket_throughput-uring-g1.svg | 60 + .../linux-socket_throughput-uring-g2.svg | 35 + .../linux-socket_throughput-uring-g3.svg | 60 + .../ROOT/images/bench/linux-summary-epoll.svg | 51 + .../ROOT/images/bench/linux-summary-uring.svg | 56 + .../ROOT/images/bench/macos-accept_churn.svg | 51 + .../ROOT/images/bench/macos-fan_out-g1.svg | 48 + .../ROOT/images/bench/macos-fan_out-g2.svg | 62 + .../ROOT/images/bench/macos-http_server.svg | 55 + .../bench/macos-local_socket_latency.svg | 60 + .../macos-local_socket_throughput-g1.svg | 62 + .../macos-local_socket_throughput-g2.svg | 62 + .../images/bench/macos-socket_latency.svg | 58 + .../bench/macos-socket_throughput-g1.svg | 60 + .../bench/macos-socket_throughput-g2.svg | 33 + .../bench/macos-socket_throughput-g3.svg | 60 + .../ROOT/images/bench/macos-summary.svg | 47 + .../images/bench/windows-accept_churn.svg | 49 + .../ROOT/images/bench/windows-fan_out-g1.svg | 50 + .../ROOT/images/bench/windows-fan_out-g2.svg | 62 + .../ROOT/images/bench/windows-http_server.svg | 57 + .../images/bench/windows-socket_latency.svg | 60 + .../bench/windows-socket_throughput-g1.svg | 60 + .../bench/windows-socket_throughput-g2.svg | 33 + .../bench/windows-socket_throughput-g3.svg | 62 + .../ROOT/images/bench/windows-summary.svg | 32 + doc/modules/ROOT/nav.adoc | 5 +- doc/modules/ROOT/pages/benchmark-report.adoc | 1194 -------------- doc/modules/ROOT/pages/benchmarks/index.adoc | 40 + doc/modules/ROOT/pages/benchmarks/linux.adoc | 1135 ++++++++++++++ doc/modules/ROOT/pages/benchmarks/macos.adoc | 595 +++++++ .../ROOT/pages/benchmarks/windows.adoc | 426 +++++ 66 files changed, 6626 insertions(+), 1211 deletions(-) create mode 100644 .github/bench/report_page.py create mode 100644 .github/workflows/benchmark-report.yml create mode 100644 bench/README.md create mode 100644 doc/modules/ROOT/images/bench/linux-accept_churn-epoll.svg create mode 100644 doc/modules/ROOT/images/bench/linux-accept_churn-uring.svg create mode 100644 doc/modules/ROOT/images/bench/linux-fan_out-epoll-g1.svg create mode 100644 doc/modules/ROOT/images/bench/linux-fan_out-epoll-g2.svg create mode 100644 doc/modules/ROOT/images/bench/linux-fan_out-uring-g1.svg create mode 100644 doc/modules/ROOT/images/bench/linux-fan_out-uring-g2.svg create mode 100644 doc/modules/ROOT/images/bench/linux-http_server-epoll.svg create mode 100644 doc/modules/ROOT/images/bench/linux-http_server-uring.svg create mode 100644 doc/modules/ROOT/images/bench/linux-local_socket_latency-epoll.svg create mode 100644 doc/modules/ROOT/images/bench/linux-local_socket_latency-uring.svg create mode 100644 doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g1.svg create mode 100644 doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g2.svg create mode 100644 doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g1.svg create mode 100644 doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g2.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_latency-epoll.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_latency-uring.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g1.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g2.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g3.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g1.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g2.svg create mode 100644 doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g3.svg create mode 100644 doc/modules/ROOT/images/bench/linux-summary-epoll.svg create mode 100644 doc/modules/ROOT/images/bench/linux-summary-uring.svg create mode 100644 doc/modules/ROOT/images/bench/macos-accept_churn.svg create mode 100644 doc/modules/ROOT/images/bench/macos-fan_out-g1.svg create mode 100644 doc/modules/ROOT/images/bench/macos-fan_out-g2.svg create mode 100644 doc/modules/ROOT/images/bench/macos-http_server.svg create mode 100644 doc/modules/ROOT/images/bench/macos-local_socket_latency.svg create mode 100644 doc/modules/ROOT/images/bench/macos-local_socket_throughput-g1.svg create mode 100644 doc/modules/ROOT/images/bench/macos-local_socket_throughput-g2.svg create mode 100644 doc/modules/ROOT/images/bench/macos-socket_latency.svg create mode 100644 doc/modules/ROOT/images/bench/macos-socket_throughput-g1.svg create mode 100644 doc/modules/ROOT/images/bench/macos-socket_throughput-g2.svg create mode 100644 doc/modules/ROOT/images/bench/macos-socket_throughput-g3.svg create mode 100644 doc/modules/ROOT/images/bench/macos-summary.svg create mode 100644 doc/modules/ROOT/images/bench/windows-accept_churn.svg create mode 100644 doc/modules/ROOT/images/bench/windows-fan_out-g1.svg create mode 100644 doc/modules/ROOT/images/bench/windows-fan_out-g2.svg create mode 100644 doc/modules/ROOT/images/bench/windows-http_server.svg create mode 100644 doc/modules/ROOT/images/bench/windows-socket_latency.svg create mode 100644 doc/modules/ROOT/images/bench/windows-socket_throughput-g1.svg create mode 100644 doc/modules/ROOT/images/bench/windows-socket_throughput-g2.svg create mode 100644 doc/modules/ROOT/images/bench/windows-socket_throughput-g3.svg create mode 100644 doc/modules/ROOT/images/bench/windows-summary.svg delete mode 100644 doc/modules/ROOT/pages/benchmark-report.adoc create mode 100644 doc/modules/ROOT/pages/benchmarks/index.adoc create mode 100644 doc/modules/ROOT/pages/benchmarks/linux.adoc create mode 100644 doc/modules/ROOT/pages/benchmarks/macos.adoc create mode 100644 doc/modules/ROOT/pages/benchmarks/windows.adoc diff --git a/.github/bench/report_page.py b/.github/bench/report_page.py new file mode 100644 index 000000000..a67efc819 --- /dev/null +++ b/.github/bench/report_page.py @@ -0,0 +1,1375 @@ +#!/usr/bin/env python3 +"""Generate the published benchmark report pages and their charts. + +Consumes the artifacts of the benchmark-report workflow (one directory per +platform holding -.json raw runs plus environment.json) and +emits a landing page plus one page per platform, and one diverging-bar SVG +per category per platform. The pages are fully generated: nothing on them +is hand-maintained. + +Metric selection and direction rules mirror .github/bench/compare.py; the +duplication is deliberate so each script stays single-file. +""" +import argparse +import json +import math +import re +import statistics +import sys +from pathlib import Path + +BASELINE = "asio_callback" +HIGHER_BETTER = ("bytes_per_sec", "items_per_sec", "ops_per_sec") +FNAME = re.compile(r"^(?P[a-z_]+(?:-[a-z]+)?)-(?P\d+)\.json$") + +# Series order and display names are fixed per platform; colors are fixed per +# entity and never cycled. Palette validated with the dataviz skill's +# validate_palette.js (light surface, 2026-09-30: all checks pass; the aqua +# contrast WARN is relieved by direct labels and the exact-value tables). +# +# On Linux, asio is built twice (epoll and io_uring reactors) so every +# comparison can pair implementations on the SAME reactor. A multi-backend +# platform's Results section charts each corosio backend on its own +# (see write_outputs/build_platform_page): one chart per category per +# backend, its series just that backend's corosio config plus its +# matched asio flavors, so a reader comparing epoll never has +# io_uring bars (or the other reactor's callback flavor) sharing the +# plot. PLATFORM_SERIES itself stays the flat per-platform list: it still +# drives the Detailed results table's row order and, on a single-backend +# platform, is used as-is for that platform's one chart per category. +PLATFORM_SERIES = { + "linux": ["corosio-uring", "corosio-epoll", "asio-uring"], + "windows": ["corosio-iocp", "asio"], + "macos": ["corosio-kqueue", "asio"], +} +SERIES_LABEL = { + "corosio-epoll": "corosio (epoll)", "corosio-uring": "corosio (io_uring)", + "corosio-iocp": "corosio (IOCP)", "corosio-kqueue": "corosio (kqueue)", + "asio_callback": "asio (callbacks)", "asio": "asio (coroutines)", + "asio_callback-epoll": "asio (callbacks, epoll)", + "asio_callback-uring": "asio (callbacks, io_uring)", + "asio-epoll": "asio (coroutines, epoll)", + "asio-uring": "asio (coroutines, io_uring)", +} +PLATFORM_DISPLAY = {"linux": "Linux", "windows": "Windows", "macos": "macOS"} +BACKEND_DISPLAY = {"epoll": "epoll", "uring": "io_uring", "iocp": "IOCP", + "kqueue": "kqueue"} + + +def platform_display(platform): + return PLATFORM_DISPLAY.get(platform, platform.capitalize()) + + +def backend_display(config): + """Return the human-readable backend name for a `corosio-` + config (e.g. 'io_uring' for 'corosio-uring', 'IOCP' for + 'corosio-iocp').""" + suffix = config[len("corosio-"):] if config.startswith("corosio-") else config + return BACKEND_DISPLAY.get(suffix, suffix) + + +def series_color(config): + if config == "corosio-uring": + return "#eb6834" + if config.startswith("corosio"): + return "#2a78d6" + return "#1baf7a" + + +def series_class(config): + if config == "corosio-uring": + return "bch-s-uring" + if config.startswith("corosio"): + return "bch-s-native" + return "bch-s-cb" + + +# SVG chart rendering constants +CHART_W = 720 +LABEL_W = 230 +LABEL_ROOM = 48 # px reserved each side of the plot for value-label text +BAR_H, BAR_GAP, GROUP_GAP = 12, 2, 10 +MARGIN_T, MARGIN_B, LEGEND_H = 36, 34, 22 +INK, INK2, GRID = "#1f1f1e", "#5f5e58", "#e3e2dc" +SURFACE = "#fcfcfb" +FONT = "font-family='-apple-system,Segoe UI,Helvetica,Arial,sans-serif'" + +# Dark counterparts of the light colors above, validated with the dataviz +# skill's validate_palette.js (dark surface, 2026-09-30: all checks pass; +# series hues carried over from light, only lightened to stay legible on +# a dark surface). +DARK_SURFACE = "#1a1a19" +DARK_INK, DARK_INK2, DARK_GRID = "#ffffff", "#c3c2b7", "#3a3a38" + +# Diverging poles for the per-platform summary chart (faster/slower), plus +# a deliberately gray, unvalidated neutral for "within noise". Poles +# validated with the dataviz skill's validate_palette.js as a 2-slot +# categorical pair: light "#2a78d6,#e34948" (surface #fcfcfb) and dark +# "#3987e5,#e66767" (surface #0d0e0f, the site's actual dark background, +# since the chart surface is transparent there) — both ALL PASS, +# 2026-09-30. +DIV_POS, DIV_POS_DARK = "#2a78d6", "#3987e5" +DIV_NEG, DIV_NEG_DARK = "#e34948", "#e66767" +DIV_MID, DIV_MID_DARK = "#f0efec", "#383835" + +# (class, property, light value, dark value); "fill" classes set fill, the +# two line classes set stroke instead. +_THEME_RULES = [ + ("bch-surface", "fill", SURFACE, DARK_SURFACE), + ("bch-ink", "fill", INK, DARK_INK), + ("bch-ink2", "fill", INK2, DARK_INK2), + ("bch-grid", "stroke", GRID, DARK_GRID), + ("bch-zero", "stroke", INK2, DARK_INK2), + ("bch-s-native", "fill", "#2a78d6", "#3987e5"), + ("bch-s-uring", "fill", "#eb6834", "#d95926"), + ("bch-s-cb", "fill", "#1baf7a", "#199e70"), + ("bch-div-pos", "fill", DIV_POS, DIV_POS_DARK), + ("bch-div-neg", "fill", DIV_NEG, DIV_NEG_DARK), + ("bch-div-mid", "fill", DIV_MID, DIV_MID_DARK), +] + + +def _rules(selector_fmt, value_index): + return "".join(f"{selector_fmt.format(cls)}{{{prop}:{vals[value_index]}}}" + for cls, prop, *vals in _THEME_RULES) + + +def _site_rules(selector_fmt, value_index): + """Like `_rules`, but bch-surface goes transparent. + + Once inlined, the chart sits directly on the page's own background, + which this SVG has no way to know the exact value of — going + transparent makes it match exactly instead of overlaying a baked-in + guess that can drift from the site's actual token. Not used for the + `prefers-color-scheme` block: a standalone-opened SVG has no page + background to match, so it keeps its own solid surface color there. + """ + parts = [] + for cls, prop, *vals in _THEME_RULES: + v = "transparent" if cls == "bch-surface" else vals[value_index] + parts.append(f"{selector_fmt.format(cls)}{{{prop}:{v}}}") + return "".join(parts) + + +# The docs site toggles dark mode with an `html.dark` class rather than +# `prefers-color-scheme`, so an -loaded SVG (which never sees the host +# page's class list) can't follow it — the chart must be inlined and carry +# its own theme-aware CSS. Cascade order matters: the media query covers a +# standalone SVG opened outside any page (no `html` ancestor to match +# against); `html:not(.dark)` then re-asserts light so a site explicitly +# set to light beats an OS set to dark; `html.dark` comes last so the +# site's own dark toggle has final say over that re-assertion. CSS always +# beats the presentation attributes on the elements themselves, so none of +# this needs `!important`, and the attributes stand as the light-mode +# fallback for renderers that ignore " +) + + +def _esc(s): + return (str(s).replace("&", "&").replace("<", "<").replace(">", ">")) + + +def _axis_bound(values): + m = max((abs(v) for v in values), default=5.0) + for b in (5, 10, 15, 20, 30, 40, 50): + if m <= b: + return float(b) + return 50.0 + + +def _format_pct(v): + """Format percentage, handling -0% → 0%.""" + s = f"{v:+.0f}%" + return "0%" if s == "-0%" else s + + +def _format_pct1(v): + """Format percentage to one decimal, handling -0.0% → 0.0%.""" + s = f"{v:+.1f}%" + return "0.0%" if s == "-0.0%" else s + + +# Conservative px/char at font-size 10-11 (real glyph widths vary and can +# run wider than a naive average estimate) — clamping against this is a +# guarantee, not an estimate: the label may end up overlapping its own +# bar or another label when clamped, but a legible, slightly-overlapping +# label beats one silently clipped off the canvas. +LABEL_CHAR_W = 8 + + +def _clamp_start(tx, text): + """Clamp an anchor='start' label so it can't run off the right edge.""" + return min(tx, CHART_W - LABEL_CHAR_W * len(text) - 2) + + +def _clamp_end(tx, text): + """Clamp an anchor='end' label so it can't run off the left edge.""" + return max(tx, LABEL_CHAR_W * len(text) + 2) + + +def _clamp_middle(cx, text): + """Clamp an anchor='middle' label so it can't run off either edge.""" + half = LABEL_CHAR_W * len(text) / 2.0 + return min(max(cx, half + 2), CHART_W - half - 2) + + +def base_family(name): + """Related-type key: the family prefix before any '/N' parameter, + with a `_lockless` suffix folded into its base so a family and its + lockless twin always land on the same chart.""" + fam = name.split("/", 1)[0] + if fam.endswith("_lockless"): + fam = fam[: -len("_lockless")] + return fam + + +MAX_CHART_GROUPS = 12 # benchmarks per chart before a category splits + + +def split_benchmarks(names): + """Partition sorted benchmark names into chart-sized runs. + + Families (see `base_family`) are never split across charts; runs + pack greedily in name order until the next family would overflow + MAX_CHART_GROUPS. A single family larger than the cap gets its own + oversized chart rather than being broken up. + """ + fams = [] + for n in names: + key = base_family(n) + if fams and fams[-1][0] == key: + fams[-1][1].append(n) + else: + fams.append((key, [n])) + parts, cur = [], [] + for _, ns in fams: + if cur and len(cur) + len(ns) > MAX_CHART_GROUPS: + parts.append(cur) + cur = [] + cur.extend(ns) + if cur: + parts.append(cur) + return parts + + +def chart_file_names(base, names): + """The SVG file name(s) `render_chart` writes for this benchmark + set: the plain `.svg` when one chart suffices, otherwise + `-g1.svg` ... in `split_benchmarks` order. Shared with the + page builder so emission never guesses at file names.""" + parts = split_benchmarks(sorted(names)) + if len(parts) <= 1: + return [f"{base}.svg"] + return [f"{base}-g{i}.svg" for i in range(1, len(parts) + 1)] + + +# Vertical-column chart geometry +VMARGIN_L = 52 # px for the y-axis percent labels +VMARGIN_R = 16 +VPLOT_H = 220 # fixed plot height; width stays CHART_W +COL_GAP = 2 # px between a group's two columns +VGROUP_GAP = 12 # px between benchmark groups +X_TICK_CHAR_W = 5.4 # ~10px-font char width, for rotated tick room + + +def render_chart(platform, category, agg_subset, series, out_base, + descs=None, backend=None): + """Write this category's vertical-column SVG chart(s). + + Columns rise (faster) or fall (slower) from a horizontal baseline + representing Boost.Asio (callbacks) on the same reactor. Categories + holding more than MAX_CHART_GROUPS benchmarks split into several + charts along related-family lines (see `split_benchmarks`), named + `-g1.svg` and so on; smaller categories keep the single + `.svg`. Returns the list of paths written. + """ + names = sorted(agg_subset) + if not names: + return [] + parts = split_benchmarks(names) + fnames = chart_file_names(out_base.name, names) + written = [] + for i, (part, fname) in enumerate(zip(parts, fnames), 1): + out = out_base.parent / fname + out.unlink(missing_ok=True) + _render_chart_part(platform, category, agg_subset, part, series, + out, descs or {}, backend, + part_no=i if len(parts) > 1 else None) + if out.exists(): + written.append(out) + return written + + +def _nice_step(span): + """A tick step from the 1/2/2.5/5 family giving ~4 intervals.""" + raw = span / 4.0 + mag = 10.0 ** math.floor(math.log10(raw)) if raw > 0 else 1.0 + for m in (1, 2, 2.5, 5, 10): + if raw <= m * mag: + return m * mag + return 10 * mag + + +def _render_chart_part(platform, category, agg_subset, names, series, + out_path, descs, backend, part_no=None): + rows = [(n, [(c, agg_subset[n]["rel"].get(c)) for c in series + if c in agg_subset[n]["rel"]]) for n in names] + rows = [(n, bars) for n, bars in rows if bars] + if not rows: + return + all_vals = [v for _, bars in rows for _, v in bars] + # Axis fitted to the data so the bars fill the plot: pad past the + # extremes, snap to a tick grid, and always include the baseline. + vmax = max(max(all_vals), 0.0) + vmin = min(min(all_vals), 0.0) + pad = max((vmax - vmin) * 0.15, 1.0) + step = _nice_step((vmax - vmin) + 2 * pad) + hi = math.ceil((vmax + pad) / step) * step + lo = math.floor((vmin - pad) / step) * step + lower_better = any(agg_subset[n]["direction"] == "lower" + for n, _ in rows) + + plot_left = VMARGIN_L + plot_right = CHART_W - VMARGIN_R + plot_w = plot_right - plot_left + top = MARGIN_T + LEGEND_H + def y(v): + return top + VPLOT_H - (min(max(v, lo), hi) - lo) \ + / (hi - lo) * VPLOT_H + y0 = y(0.0) # the baseline + + n = len(rows) + gw = (plot_w - VGROUP_GAP * (n - 1)) / n + k = max(len(bars) for _, bars in rows) + col_w = min(26.0, (gw - COL_GAP * (k - 1)) / k) + group_pad = (gw - (col_w * k + COL_GAP * (k - 1))) / 2.0 + + # Rotated tick labels need vertical room proportional to the longest + # benchmark name; sin(38 deg) ~ 0.62. + max_chars = max(len(nm) for nm, _ in rows) + tick_h = min(130, max(36, int(0.62 * max_chars * X_TICK_CHAR_W) + 20)) + height = top + VPLOT_H + tick_h + 10 + label_all = n <= 8 + ext = {} + for _, bars in rows: + for c, v in bars: + elo, ehi = ext.get(c, (v, v)) + ext[c] = (min(elo, v), max(ehi, v)) + + # The visible title lives in the page heading above the chart; this + # is the SVG's accessible name and must be the first child. + # The backend is named once in the hint line; series and baseline + # labels stay backend-free so the legend doesn't repeat it. + bdisp = backend or next( + (backend_display(c) for c in series if c.startswith("corosio")), None) + plat = _esc(platform_display(platform)) + title = (f"{_esc(category)} ({_esc(backend)}) — {plat}" if backend + else f"{_esc(category)} — {plat}") + if part_no: + title += f", part {part_no}" + s = [f"<svg xmlns='http://www.w3.org/2000/svg' class='bch-chart' " + f"width='{CHART_W}' height='{height:.0f}' " + f"viewBox='0 0 {CHART_W} {height:.0f}'>", + f"<title>{title}", + CHART_STYLE, + f"", + f"↑ faster" + + (" (lower mean latency)" if lower_better else "") + + " · dotted line = asio (callbacks)" + + (f" · backend = {_esc(bdisp)}" if bdisp else "") + + ""] + lx = 12 + for c in series: + s.append(f"") + label = "corosio" if c.startswith("corosio") else "asio (coroutines)" + s.append(f"{_esc(label)}") + lx += 14 + 7 * len(label) + 18 + # horizontal gridlines and percent labels at each tick (0 gets the + # dotted baseline below instead of a plain gridline) + gv = lo + while gv <= hi + 1e-9: + if abs(gv) > 1e-9: + gy = y(gv) + s.append(f"") + s.append(f"{gv:+g}%") + gv += step + s.append(f"0%") + + for i, (name, bars) in enumerate(rows): + gx = plot_left + i * (gw + VGROUP_GAP) + desc = descs.get(name) + cx = gx + group_pad + for c, v in bars: + yv = y(v) + floor = top + VPLOT_H + rect = (f"{_esc(name)} — " + f"{_esc(desc)}") + else: + s.append(f"{rect}/>") + if label_all or v in ext[c]: + text = _format_pct(v) + ty = yv - 4 + tx = _clamp_middle(cx + col_w / 2.0, text) + s.append(f"{text}") + cx += col_w + COL_GAP + tx_, ty_ = gx + gw / 2.0, top + VPLOT_H + 14 + s.append(f"" + f"{_esc(name)}") + + # The baseline goes on top of the bars: a threshold must stay + # readable across the full width, not vanish behind whatever + # column happens to cross it. + s.append(f"") + s.append("") + out_path.parent.mkdir(parents=True, exist_ok=True) + out_path.write_text("\n".join(s) + "\n") + + +SEG_GAP = 2 # px gap between adjacent stacked-bar segments +SEG_LABEL_MIN_W = 16 # min segment width (px) to show its count label +MARGIN_R = 70 # px reserved right of the bar for the median annotation + + +def render_summary_chart(platform, summary, out_path, backend=None): + """Write one platform's diverging stacked-bar summary SVG. + + One row per category (sorted by descending median rel for this + config), each a single bar splitting its benchmark count into slower/ + within-noise/faster segments that straddle a shared center line. All + rows share one px-per-benchmark scale (set by the category with the + most benchmarks) so segment lengths are comparable down the page. + Skipped (no file) when there's nothing to summarize. + + `backend` is the display name (e.g. "epoll") for a multi-backend + platform's per-backend chart; omitted for a single-backend platform's + one chart, which needs no backend qualifier in its title. + """ + if not summary: + return + rows = sorted(summary.items(), key=lambda kv: kv[1]["median"], reverse=True) + + plot_w = CHART_W - LABEL_W - 20 + x0 = LABEL_W + plot_w / 2.0 # center line + half_w = plot_w / 2.0 + # Layout diverges from the CENTER, so each side only ever has half_w + # to work with, not the full plot_w a one-sided scale would assume; + # sizing against max_total (as if bars started at x=0) let the widest + # one-sided bar run off the right edge. max_extent is that widest + # single-side reach (neutral counts split in half since they straddle + # center); MARGIN_R reserves room for the median annotation that sits + # to the right of the bar's actual right edge. + max_extent = max(max(c["slower"] + c["within"] / 2.0, + c["faster"] + c["within"] / 2.0) for _, c in rows) + px = (half_w - MARGIN_R) / max_extent if max_extent else 0.0 + + plot_h = len(rows) * BAR_H + GROUP_GAP * (len(rows) - 1) + height = MARGIN_T + plot_h + MARGIN_B + + # See render_chart's comment: the visible title moved to an HTML + # heading on the page, and this — the SVG's own accessible + # name — must be the first child of <svg>, before <style>. + plat = _esc(platform_display(platform)) + title = f"Summary ({backend}) — {plat}" if backend else f"Summary — {plat}" + s = [f"<svg xmlns='http://www.w3.org/2000/svg' class='bch-chart' " + f"width='{CHART_W}' height='{height:.0f}' " + f"viewBox='0 0 {CHART_W} {height:.0f}'>", + f"<title>{title}", + CHART_STYLE, + f"", + f"← slower · within noise · faster " + f"→ per benchmark, vs Boost.Asio (callbacks)"] + + top = MARGIN_T + s.append(f"") + + y = top + for category, cat in rows: + s.append(f"{_esc(category)}") + + # Neutral straddles center; slower/faster extend outward from its + # edges, each offset by SEG_GAP. This falls out naturally even + # when within=0: the two gaps collapse to one empty slot at center. + w_neg, w_mid, w_pos = (cat["slower"] * px, cat["within"] * px, + cat["faster"] * px) + mid_l, mid_r = x0 - w_mid / 2.0, x0 + w_mid / 2.0 + neg_r = mid_l - SEG_GAP + neg_l = neg_r - w_neg + pos_l = mid_r + SEG_GAP + pos_r = pos_l + w_pos + + for x1_, x2_, cls, fill, count, label_cls, label_fill, verdict in ( + (neg_l, neg_r, "bch-div-neg", DIV_NEG, cat["slower"], + "bch-seg-label", "#ffffff", "slower"), + (mid_l, mid_r, "bch-div-mid", DIV_MID, cat["within"], + "bch-ink", INK, "within noise"), + (pos_l, pos_r, "bch-div-pos", DIV_POS, cat["faster"], + "bch-seg-label", "#ffffff", "faster"), + ): + w = x2_ - x1_ + if w <= 0: + continue + title = (f"{count} benchmark{'s' if count != 1 else ''} " + f"{verdict} than asio (callbacks)") + s.append(f"" + f"{_esc(title)}") + if w >= SEG_LABEL_MIN_W: + cx = _clamp_middle((x1_ + x2_) / 2.0, str(count)) + s.append(f"" + f"{count}") + + median_text = _format_pct1(cat["median"]) + median_x = _clamp_start(pos_r + 4, median_text) + s.append(f"{median_text}") + y += BAR_H + GROUP_GAP + + s.append("") + out_path.parent.mkdir(parents=True, exist_ok=True) + out_path.write_text("\n".join(s) + "\n") + + +def primary_metric(category, metrics): + """Pick the compared metric and its direction for one benchmark.""" + if "latency" in category and "latency_mean_ns" in metrics: + return "latency_mean_ns", "lower" + for m in HIGHER_BETTER: + if m in metrics: + return m, "higher" + if "latency_mean_ns" in metrics: + return "latency_mean_ns", "lower" + return None, None + + +def load_platform(platform_dir): + """Load one platform's raw runs, descriptions, and environment.json. + + Descriptions are corosio-only (the asio-side binaries don't emit + them): per-benchmark `"description"` strings and a top-level + `"category_descriptions"` object, both optional. The numeric-only + filter below that builds each benchmark's metric table would silently + drop the string description field, so it's captured on the side + instead, first-non-empty-wins across whichever configs/iterations + have it — returned as a third element rather than folded into `runs`, + since it's page-rendering metadata, not a metric. + """ + runs, env = {}, None + bench_descs, cat_descs = {}, {} + p_env = Path(platform_dir) / "environment.json" + if p_env.exists(): + try: + env = json.loads(p_env.read_text()) + except (OSError, json.JSONDecodeError) as e: + print(f"warning: unreadable environment.json: {e}", file=sys.stderr) + for p in sorted(Path(platform_dir).iterdir()): + m = FNAME.match(p.name) + if not m: + continue + config, it = m.group("config"), int(m.group("iter")) + try: + payload = json.loads(p.read_text()) + except (OSError, json.JSONDecodeError) as e: + print(f"warning: skipping unreadable {p.name}: {e}", file=sys.stderr) + continue + if not isinstance(payload, dict): + print(f"warning: skipping non-object payload {p.name}", file=sys.stderr) + continue + benches = payload.get("benchmarks", []) + if not isinstance(benches, list): + print(f"warning: skipping non-list benchmarks in {p.name}", file=sys.stderr) + continue + cds = payload.get("category_descriptions") + if isinstance(cds, dict): + for cat, desc in cds.items(): + if isinstance(desc, str) and desc and cat not in cat_descs: + cat_descs[cat] = desc + table = {} + for b in benches: + if not isinstance(b, dict): + print(f"warning: skipping non-object entry in {p.name}", file=sys.stderr) + continue + key = (b.get("category", ""), b.get("name", "")) + table[key] = {k: v for k, v in b.items() + if isinstance(v, (int, float)) and not isinstance(v, bool)} + desc = b.get("description") + if isinstance(desc, str) and desc and key not in bench_descs: + bench_descs[key] = desc + runs.setdefault(config, {})[it] = table + return runs, env, {"benchmarks": bench_descs, "categories": cat_descs} + + +def _cv(vals): + if len(vals) < 2: + return None + mean = statistics.fmean(vals) + if mean == 0: + return None + return statistics.stdev(vals) / abs(mean) * 100.0 + + +def aggregate(runs): + """Median + CV of the primary metric per benchmark per configuration.""" + keys, sample = set(), {} + for config, iters in runs.items(): + for table in iters.values(): + keys.update(table) + for k, metrics in table.items(): + sample.setdefault(k, metrics) + agg = {} + for key in sorted(keys): + category, _ = key + metric, direction = primary_metric(category, sample.get(key, {})) + if metric is None: + continue + configs = {} + for config, iters in runs.items(): + vals = [t[key][metric] for _, t in sorted(iters.items()) + if key in t and metric in t[key]] + if not vals: + continue + configs[config] = {"median": statistics.median(vals), + "cv": _cv(vals), "n": len(vals)} + if configs: + agg[key] = {"metric": metric, "direction": direction, + "configs": configs} + return agg + + +def baseline_for(config, present_configs): + """Return the asio baseline config matched to `config`, or None if + `config` IS itself a baseline (gets no rel of its own). + + On Linux, asio is built once per reactor (epoll, io_uring), so a + `corosio-` or `asio_callback-` config compares + against the SAME-reactor asio build (`asio-`) when one is + present, rather than a single reactor-blind `asio`. Configs with no + reactor suffix — single-reactor platforms, or a dataset that only + ever shipped plain `asio`/`asio_callback` — fall back to plain + `asio`, exactly as before this reactor-matching was added. + """ + if config == BASELINE or config.startswith("asio_callback-"): + return None + if "-" in config: + prefix, _, flavor = config.rpartition("-") + if prefix in ("corosio", "asio"): + matched = f"asio_callback-{flavor}" + if matched in present_configs: + return matched + return BASELINE + + +def _configs_seen(agg): + """All configs with at least one benchmark's data anywhere in `agg`.""" + return {c for row in agg.values() for c in row["configs"]} + + +def corosio_configs_present(agg, series): + """Which of `series`'s corosio-* configs actually have data in `agg`. + + `series` (PLATFORM_SERIES[platform]) is a static per-platform list; + on Linux it names both reactors even when a given run only actually + captured one of them (e.g. an older-shaped dataset, or a partial + artifact), so "is this platform multi-backend" must be answered from + the real data, not the static list. + """ + seen = _configs_seen(agg) + return [c for c in series if c.startswith("corosio") and c in seen] + + +def relativize(agg): + """Add sign-normalized percent-vs-matched-baseline for every config + that has one (see `baseline_for`); baseline configs get no rel.""" + for row in agg.values(): + present = row["configs"] + rel = {} + for config, c in present.items(): + baseline_key = baseline_for(config, present) + if baseline_key is None: + continue + base = present.get(baseline_key) + if not base or base["median"] <= 0: + continue + r = (c["median"] / base["median"] - 1.0) * 100.0 + if row["direction"] == "lower": + r = -r + rel[config] = r + if rel: + row["rel"] = rel + return agg + + +def _ratio_noise(cv_a, cv_b): + parts = [c for c in (cv_a, cv_b) if c is not None] + if not parts: + return None + return math.hypot(*parts) if len(parts) == 2 else parts[0] + + +def _config_rels(agg, config): + """Yield (category, name, rel_value, noise) for one config's matched- + baseline comparisons (see `baseline_for`); benchmarks where `config` + has no rel (no baseline, or `config` absent) are skipped. Shared by + `summarize()` (per-category breakdown) and the platform page's + per-backend stat line (flattened across all categories). + """ + for (category, name), row in agg.items(): + rel = row.get("rel") + if not rel or config not in rel: + continue + # Noise must pair against the SAME baseline `config` was + # relativized against, not a fixed global 'asio' CV, or a + # reactor-matched comparison would silently mix noise from an + # unrelated build. + baseline_key = baseline_for(config, row["configs"]) + base_cv = (row["configs"][baseline_key]["cv"] + if baseline_key in row["configs"] else None) + noise = _ratio_noise(row["configs"][config]["cv"], base_cv) + yield category, name, rel[config], noise + + +def summarize(agg, config): + """Per-category verdict counts for ONE corosio config vs its own + matched baseline (not an aggregate across every corosio backend — + each backend gets its own independent summary on the platform page). + """ + out = {} + vals = {} + for category, name, val, noise in _config_rels(agg, config): + cat = out.setdefault(category, {"faster": 0, "within": 0, "slower": 0}) + vals.setdefault(category, []).append(val) + if noise is not None and abs(val) <= 2.0 * noise: + cat["within"] += 1 + elif val > 0: + cat["faster"] += 1 + else: + cat["slower"] += 1 + for category, cat in out.items(): + cat["median"] = statistics.median(vals[category]) + return out + + +ENV_FIELDS = [("cpu", "CPU"), ("cores", "Cores"), ("ram_gb", "RAM (GB)"), + ("os", "OS"), ("kernel", "Kernel/build"), ("compiler", "Compiler"), + ("cmake", "CMake"), ("liburing", "liburing"), + ("boost_sha", "Boost commit"), ("asio_sha", "Asio commit"), + ("asio_reactor", "Asio reactor"), + ("capy_sha", "Capy commit"), ("corosio_sha", "Corosio commit"), + ("corosio_ref", "Corosio branch"), + ("date_utc", "Date (UTC)"), ("iterations", "Iterations"), + ("duration_s", "Duration per benchmark (s)")] + + +def human(value, metric): + """Format a metric value with readable units. + + Copied verbatim from `compare.py`'s `human()`; see this module's + docstring for why the metric-formatting logic is duplicated instead + of shared. + """ + if metric.endswith("_ns"): + for factor, unit in ((1e9, "s"), (1e6, "ms"), (1e3, "µs")): + if abs(value) >= factor: + return f"{value / factor:.2f} {unit}" + return f"{value:.0f} ns" + # Bytes glue the prefix to the unit (KB/s, GB/s); count-style metrics + # glue it to the number (1.82K ops/s) so sub-1000 values still carry a + # unit instead of a bare "/s". + if metric == "bytes_per_sec": + for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")): + if abs(value) >= factor: + n = value / factor + return (f"{n:.2f} {prefix}B/s" if n < 100 + else f"{n:.1f} {prefix}B/s") + return f"{value:.1f} B/s" + unit = "items/s" if metric == "items_per_sec" else "ops/s" + for factor, prefix in ((1e9, "G"), (1e6, "M"), (1e3, "K")): + if abs(value) >= factor: + n = value / factor + return (f"{n:.2f}{prefix} {unit}" if n < 100 + else f"{n:.1f}{prefix} {unit}") + return f"{value:.1f} {unit}" + + +def _table_rows(rows, series, names=None, configs=None): + """Render one AsciiDoc table row per (benchmark, configuration). + + Row order: the charted `series` first, then any other non-baseline + config present, then the baseline config(s) last. `names` restricts + the benchmarks (one chart's worth) and `configs` the configurations + (one backend's slice); both default to everything in `rows`. + """ + lines = [] + for name in (sorted(rows) if names is None else names): + row = rows[name] + rel = row.get("rel") or {} + present = row["configs"] + baselines = sorted(c for c in present if baseline_for(c, present) is None) + others = sorted(c for c in present + if c not in series and c not in baselines) + for config in (*series, *others, *baselines): + c = present.get(config) + if c is None or (configs is not None and config not in configs): + continue + cv = "—" if c["cv"] is None else f"{c['cv']:.2f}%" + if config in baselines: + # A flavored baseline (asio-epoll/asio-uring) notes which + # reactor it was built against; the single reactor-blind + # 'asio' baseline needs no qualifier. + if config.startswith("asio_callback-"): + flavor = config[len("asio_callback-"):] + flavor = "io_uring" if flavor == "uring" else flavor + vs = f"baseline ({flavor})" + else: + vs = "baseline" + else: + r = rel.get(config) + vs = "no asio equivalent" if r is None else f"{r:+.1f}%" + lines.append(f"| `{name}` | {SERIES_LABEL.get(config, config)} " + f"| {human(c['median'], row['metric'])} | {cv} | {vs}") + return lines + + +def _suite_size_sentence(platforms): + """Summarize benchmark/category counts per platform, from agg/by_cat.""" + clauses = [] + for platform, data in platforms.items(): + if data is None: + continue + n, c = len(data["agg"]), len(data["by_cat"]) + clauses.append(f"{n} benchmark{'s' if n != 1 else ''} across " + f"{c} categor{'ies' if c != 1 else 'y'} on " + f"{platform_display(platform)}") + if not clauses: + return None + return "This run measured " + "; ".join(clauses) + "." + + +def _generated_stamp(meta, note): + """Build the '_Generated ...' byline, omitting the run link when absent.""" + if meta["run_url"]: + source = (f" from corosio `{meta['corosio_sha'][:12]}` — " + f"{meta['run_url']}[benchmark run].") + else: + source = f" from corosio `{meta['corosio_sha'][:12]}`." + return (f"_Generated {meta['date']}{source} This page is fully " + f"generated; do not edit by hand (see {note})._") + + +def build_index_page(platforms, meta): + """Build the Benchmarks landing page: methodology + links to platforms. + + `:page-aliases:` keeps the old single-page URL resolving here, since + this replaces `benchmark-report.adoc`. + """ + L = [] + a = L.append + a("= Boost.Corosio Performance Benchmarks") + a(":page-aliases: benchmark-report.adoc") + a(":page-mode: explanation") + a(":toc: left") + a("") + a(_generated_stamp(meta, "<>")) + a("") + a("This report compares Boost.Corosio against Boost.Asio using " + "coroutines (the `asio` configuration) and Boost.Asio using " + "callback-based handlers (`asio_callback`) across the corosio " + "benchmark suite. Its purpose is to track corosio's performance " + "relative to an established async I/O library as both evolve, not " + "to crown a universal winner.") + a("") + a("== Methodology") + a("") + suite_sentence = _suite_size_sentence(platforms) + if suite_sentence: + a(suite_sentence) + a("") + a("Each benchmark runs several iterations of a fixed duration, " + "interleaved across configurations. Drift from thermal throttling " + "or background load therefore lands on every configuration alike, " + "rather than skewing one of them. The reported figure for a " + "benchmark/configuration pair is the median across its iterations, " + "which resists a single outlier iteration. The coefficient of " + "variation (CV) across those iterations appears alongside it as a " + "measure of run-to-run noise. See each platform page's Test " + "Environment section for the exact iteration count and duration " + "used there.") + a("") + a("Metric selection and the higher/lower-is-better direction follow " + "the same rules as the PR comparison bot " + "(`.github/bench/compare.py`). Latency categories compare " + "`latency_mean_ns` (lower is better); everything else compares " + "`bytes_per_sec`, then `items_per_sec`, then `ops_per_sec` (higher " + "is better). Percentages on this page are sign-normalized so that " + "a positive value always means corosio performed better than " + "asio.") + a("") + a("Every percentage on these pages is measured against Boost.Asio " + "with callback handlers on the same reactor. Boost.Asio's own " + "coroutine frontend (the `asio` configuration) is plotted " + "alongside corosio against that same baseline.") + a("") + a("On Linux, asio is built twice: once against the epoll reactor, " + "once against io_uring. Every comparison pairs implementations on " + "the same reactor, never judging an io_uring corosio backend " + "against an epoll-only asio. On Windows and macOS, asio uses its " + "native reactor (IOCP and kqueue respectively), matching corosio's " + "own default backend there. Each platform page records the reactor " + "used on that machine.") + a("") + a("Each platform's numbers come from a dedicated, otherwise-idle " + "machine, to keep competing workloads from adding noise to the " + "measurements.") + a("") + a("== Reproducing") + a("") + a("The full suite layout and exact commands are documented in " + "`bench/README.md`. In short: dispatch the benchmark-report " + "workflow, download its artifacts, and regenerate this page with:") + a("") + a("[source,bash,role=external]") + a("----") + a("python3 .github/bench/report_page.py --input-dir \\") + a(" --output-dir doc/modules/ROOT") + a("----") + a("") + if meta["run_url"]: + a(f"Source data: {meta['run_url']}[workflow run]. Regenerate this " + f"page with the command above. GitHub retains the run's raw " + f"JSON artifacts for 90 days; these pages are the durable " + f"record.") + else: + a("Regenerate this page with the command above. GitHub retains a " + "workflow run's raw JSON artifacts for 90 days; these pages " + "are the durable record.") + a("") + a("== Platforms") + a("") + for platform in platforms: + a(f"* xref:benchmarks/{platform}.adoc[{platform_display(platform)}]") + return "\n".join(L) + "\n" + + +def _chart_parts(svg_names, base): + """The actual chart file(s) on disk for `base`: the single + `.svg`, or the `-g1`... run a split category produced.""" + if f"{base}.svg" in svg_names: + return [f"{base}.svg"] + out, i = [], 1 + while f"{base}-g{i}.svg" in svg_names: + out.append(f"{base}-g{i}.svg") + i += 1 + return out + + +def build_platform_page(platform, data, generated, meta): + """Build one platform's page: summary, test environment, results.""" + L = [] + a = L.append + a(f"= {platform_display(platform)} Benchmarks") + a(":page-mode: explanation") + a(":toc: left") + a("") + a(_generated_stamp( + meta, "the xref:benchmarks/index.adoc[Benchmarks landing page]")) + a("") + if data is None: + a(f"No results for {platform} in this run.") + return "\n".join(L) + "\n" + a("== Summary") + a("") + a("_Within noise_ means the relative difference is within twice the " + "combined run-to-run noise (the root-sum-square of each side's " + "CV). Differences that small are indistinguishable from " + "measurement jitter. See xref:benchmarks/index.adoc[Methodology] " + "for how these figures are computed.") + a("") + series = PLATFORM_SERIES.get(platform, []) + corosio_configs = corosio_configs_present(data["agg"], series) + multi_backend = len(corosio_configs) > 1 + summaries = data.get("summaries") or {} + for config in corosio_configs: + summary = summaries.get(config) + if not summary: + continue + if multi_backend: + a(f"=== {backend_display(config)}") + a("") + faster = sum(cat["faster"] for cat in summary.values()) + within = sum(cat["within"] for cat in summary.values()) + slower = sum(cat["slower"] for cat in summary.values()) + total = faster + within + slower + overall = statistics.median( + [v for _, _, v, _ in _config_rels(data["agg"], config)]) + a(f"**{faster} faster · {within} within noise · " + f"{slower} slower** of {total} " + f"benchmark{'s' if total != 1 else ''} — median " + f"**{_format_pct1(overall)}** vs Boost.Asio (callbacks) on the " + f"same reactor.") + a("") + suffix = config[len("corosio-"):] + fname = (f"{platform}-summary-{suffix}.svg" if multi_backend + else f"{platform}-summary.svg") + a(f"image::bench/{fname}[summary,role=bch-block,opts=inline]") + a("") + a("== Test Environment") + a("") + env = data["env"] or {} + a('[cols="1,3"]') + a("|===") + for key, label in ENV_FIELDS: + val = env.get(key, "unknown") + # Literal tool output reads as code, and spares the spell checker + if key == "cmake": + val = f"`{val}`" + a(f"| {label} | {val}") + a("|===") + a("") + a("== Results") + a("") + svg_names = {p.name for p in generated} + series = PLATFORM_SERIES.get(platform, []) + # The chart's own visible title moved out of the SVG (now just an + # accessible-name , not rendered), so the heading is the only + # place the category name shows — every category gets one, including + # a chartless one (no benchmark in it has an asio baseline to compare + # against), which now reads identically to the rest apart from the + # missing image. + descs = data.get("descs") or {"benchmarks": {}, "categories": {}} + categories = sorted(data["by_cat"].items()) + for i, (category, rows) in enumerate(categories): + a(f"=== `{category}`") + a("") + cat_desc = descs["categories"].get(category) + if cat_desc: + # Plain AsciiDoc text, not XML/HTML-escaped: _esc() is for the + # SVG <title> elements below, and would corrupt a literal '&' + # in hand-authored description text into a visible "&". + a(cat_desc) + a("") + all_configs = {c for r in rows.values() for c in r["configs"]} + covered = set() + def emit_table(part_names, configs): + a(".Detailed results") + # bch-exact role lets the chart's CSS patch this + # collapsible's summary label color, which the theme's own + # dark rules miss + a("[%collapsible.bch-exact]") + a("====") + a('[%autowidth.stretch,options="header"]') + a("|===") + a("| Benchmark | Implementation | Median | CV | vs asio") + L.extend(_table_rows(rows, series, part_names, configs)) + a("|===") + a("====") + a("") + if multi_backend: + # One chart per corosio backend, each isolating that + # backend's own bars (and its matched asio flavor) so epoll + # and io_uring numbers are never read off the same plot; a + # bold label stands in for a heading since these are + # siblings within one category, not sections of their own. + # Each chart carries its own collapsible with exactly the + # benchmarks and configurations it plots (plus their + # baseline). + for config in corosio_configs: + suffix = config[len("corosio-"):] + fnames = _chart_parts( + svg_names, f"{platform}-{category}-{suffix}") + if not fnames: + continue + bseries = [config, f"asio-{suffix}"] + bconfigs = set(bseries) | {f"asio_callback-{suffix}"} + part_lists = split_benchmarks( + [n for n in sorted(rows) + if (rows[n].get("rel") or {}).keys() & set(bseries)]) + if part_lists: + covered.update(*part_lists) + label = backend_display(config) + a(f"**{label}**") + a("") + for part, (fname, pnames) in enumerate( + zip(fnames, part_lists), 1): + alt = f"{category.replace('_', ' ')} comparison ({label})" + if len(fnames) > 1: + alt += f", part {part}" + a("[.bch-card]") + a("--") + a(f"image::bench/{fname}[{alt},role=bch-block," + f"opts=inline]") + a("") + emit_table(pnames, bconfigs) + a("--") + a("") + else: + # inline (not <img>) so the SVG's <style> can see the docs + # site's html.dark class and follow its theme toggle; no + # fixed width here — the chart's own CSS makes it fluid. + # role=bch-block tags the wrapping <div class="imageblock"> + # so the chart's CSS can stretch just this block, not every + # imageblock on the page. + fnames = _chart_parts(svg_names, f"{platform}-{category}") + part_lists = split_benchmarks( + [n for n in sorted(rows) if rows[n].get("rel")]) + if part_lists: + covered.update(*part_lists) + for part, (fname, pnames) in enumerate( + zip(fnames, part_lists), 1): + alt = f"{category.replace('_', ' ')} comparison" + if len(fnames) > 1: + alt += f", part {part}" + a("[.bch-card]") + a("--") + a(f"image::bench/{fname}[{alt},role=bch-block," + f"opts=inline]") + a("") + emit_table(pnames, None) + a("--") + a("") + # Benchmarks no chart plots (nothing to compare against the + # baseline) still get their exact values recorded. + leftover = [n for n in sorted(rows) if n not in covered] + if leftover: + a("[.bch-card]") + a("--") + emit_table(leftover, None) + a("--") + a("") + if i < len(categories) - 1: + # Plain vertical space between categories, not a rule: <hr> + # (what a thematic break ''' becomes) renders badly in dark + # mode, so a passthrough <br/> stands in for it instead. + a("++++") + a("<br/>") + a("++++") + a("") + return "\n".join(L) + "\n" + + +def main(argv=None): + ap = argparse.ArgumentParser(prog="report_page.py") + ap.add_argument("--input-dir", required=True) + ap.add_argument("--output-dir", required=True) + ap.add_argument("--expect", default="linux,windows,macos") + args = ap.parse_args(argv) + platforms = {} + for platform in args.expect.split(","): + d = Path(args.input_dir) / f"bench-report-{platform}" + platforms[platform] = load_platform(d) if d.is_dir() else None + if platforms[platform] is None: + print(f"warning: no artifact for {platform}", file=sys.stderr) + generated = write_outputs(platforms, Path(args.output_dir)) + print(f"wrote {len(generated['pages'])} pages and " + f"{len(generated['svgs'])} charts") + return 0 + + +def write_outputs(platforms, output_dir): + pages = Path(output_dir) / "pages" / "benchmarks" + images = Path(output_dir) / "images" / "bench" + pages.mkdir(parents=True, exist_ok=True); images.mkdir(parents=True, exist_ok=True) + svgs = [] + prepared = {} + meta = {"date": "unknown", "corosio_sha": "unknown", "run_url": ""} + seen_env = False + for platform, loaded in platforms.items(): + if loaded is None: + prepared[platform] = None + continue + runs, env, descs = loaded + if env and not seen_env: + meta = {"date": env.get("date_utc", "unknown"), + "corosio_sha": env.get("corosio_sha", "unknown"), + "run_url": env.get("run_url", "")} + seen_env = True + agg = relativize(aggregate(runs)) + series = PLATFORM_SERIES.get(platform, []) + # Known once per platform, ahead of both the per-category charts + # and the per-backend summaries below: which corosio backends + # actually have data (not just which PLATFORM_SERIES names), and + # every config with any data at all (used to decide whether a + # backend's matched asio_callback flavor has anything to chart). + corosio_configs = corosio_configs_present(agg, series) + multi_backend = len(corosio_configs) > 1 + seen_configs = _configs_seen(agg) + by_cat = {} + for (category, name), row in agg.items(): + by_cat.setdefault(category, {})[name] = row + for category, rows in sorted(by_cat.items()): + cat_bench_descs = {n: descs["benchmarks"][(category, n)] + for n in rows + if (category, n) in descs["benchmarks"]} + if multi_backend: + # One chart per corosio backend rather than one shared + # chart: each backend's plot carries only its own bars + # plus its matched asio_callback flavor (omitted if that + # flavor has no data at all), so epoll and io_uring never + # share an axis or a legend. + for config in corosio_configs: + suffix = config[len("corosio-"):] + asio_cfg = f"asio-{suffix}" + backend_series = [config] + ( + [asio_cfg] if asio_cfg in seen_configs else []) + chartable = {n: r for n, r in rows.items() + if r.get("rel") and + any(c in r["rel"] for c in backend_series)} + svgs.extend(render_chart( + platform, category, chartable, backend_series, + images / f"{platform}-{category}-{suffix}", + cat_bench_descs, + backend=backend_display(config))) + else: + chartable = {n: r for n, r in rows.items() if r.get("rel")} + svgs.extend(render_chart( + platform, category, chartable, series, + images / f"{platform}-{category}", cat_bench_descs)) + # One independent summary (counts + chart) per corosio backend — + # on Linux that's epoll and io_uring separately, not a single + # "best of the two" aggregate; single-backend platforms still get + # exactly one, just without a backend-suffixed filename. + summaries = {} + for config in corosio_configs: + summary = summarize(agg, config) + suffix = config[len("corosio-"):] + fname = (f"{platform}-summary-{suffix}.svg" if multi_backend + else f"{platform}-summary.svg") + out = images / fname + out.unlink(missing_ok=True) # same stale-resurrection guard as above + render_summary_chart(platform, summary, out, + backend=backend_display(config) + if multi_backend else None) + if out.exists(): + svgs.append(out) + summaries[config] = summary + prepared[platform] = {"env": env, "agg": agg, "by_cat": by_cat, + "summaries": summaries, "descs": descs} + for stale in images.glob("*.svg"): + if stale not in svgs: + stale.unlink() + + out_pages = [pages / "index.adoc"] + out_pages[0].write_text(build_index_page(prepared, meta)) + for platform, data in prepared.items(): + page = pages / f"{platform}.adoc" + page.write_text(build_platform_page(platform, data, svgs, meta)) + out_pages.append(page) + + emitted_names = {p.name for p in out_pages} + for stale in pages.glob("*.adoc"): + if stale.name not in emitted_names: + stale.unlink() + + return {"pages": out_pages, "svgs": svgs} + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/benchmark-report.yml b/.github/workflows/benchmark-report.yml new file mode 100644 index 000000000..c6803b816 --- /dev/null +++ b/.github/workflows/benchmark-report.yml @@ -0,0 +1,229 @@ +# +# Copyright (c) 2026 Steve Gerbino +# +# Distributed under the Boost Software License, Version 1.0. (See accompanying +# file LICENSE_1_0.txt or copy at http://www.boost.org/LICENSE_1_0.txt) +# +# Official repository: https://github.com/cppalliance/corosio/ +# +# Comparison benchmark run for the published documentation report. +# Dispatch-only; artifacts are turned into doc pages locally — see +# bench/README.md ("Refreshing the published report"). +# +# Runner prerequisites: same as benchmarks.yml. + +name: benchmark-report + +on: + workflow_dispatch: + inputs: + iterations: + description: "iterations per configuration" + default: "7" + duration: + description: "seconds per benchmark" + default: "2" + +permissions: + contents: read + +concurrency: + group: benchmark-report + cancel-in-progress: false + +jobs: + bench: + strategy: + fail-fast: false + matrix: + include: + - platform: linux + labels: '["self-hosted", "Linux", "X64"]' + configs: "corosio-epoll corosio-uring asio-epoll asio-uring asio_callback-epoll asio_callback-uring" + - platform: windows + labels: '["self-hosted", "Windows", "X64"]' + configs: "corosio-iocp asio asio_callback" + - platform: macos + labels: '["self-hosted", "macOS", "ARM64"]' + configs: "corosio-kqueue asio asio_callback" + name: bench-report-${{ matrix.platform }} + runs-on: ${{ fromJSON(matrix.labels) }} + timeout-minutes: 240 + env: + ITERS: ${{ inputs.iterations || '7' }} + DUR: ${{ inputs.duration || '2' }} + defaults: + run: + shell: bash + steps: + - name: Clean workspace + run: rm -rf results build-aoff build-aon ws boost-root && mkdir -p results + + - name: Checkout corosio + uses: actions/checkout@v4 + with: + path: ws/corosio + persist-credentials: false + + - name: Resolve Capy branch + id: capy-ref + uses: ./ws/corosio/.github/actions/resolve-capy + + - name: Checkout capy + uses: actions/checkout@v4 + with: + repository: ${{ steps.capy-ref.outputs.repo }} + ref: ${{ steps.capy-ref.outputs.ref }} + path: ws/capy + persist-credentials: false + + - name: Clone Boost + uses: alandefreitas/cpp-actions/boost-clone@v1.9.0 + with: + branch: develop + boost-dir: boost-root + modules: asio + modules-exclude-paths: '' + scan-modules-dir: ws/corosio + scan-modules-ignore: corosio, capy + + - name: Patch Boost + run: | + set -e + rm -rf boost-root/libs/corosio boost-root/libs/capy + cp -r ws/corosio boost-root/libs/corosio + cp -r ws/capy boost-root/libs/capy + + - name: Put common toolchain dirs on PATH + run: | + for dir in /opt/homebrew/bin /usr/local/bin "$HOME/.local/bin" /snap/bin /opt/cmake/bin; do + if [ -d "$dir" ]; then echo "$dir" >> "$GITHUB_PATH"; fi + done + + - name: Configure and build (epoll/default asio reactor) + run: | + set -e + cmake -S boost-root -B build-aoff \ + -DCMAKE_BUILD_TYPE=Release \ + -DBOOST_INCLUDE_LIBRARIES="corosio;asio" \ + -DBOOST_COROSIO_BUILD_BENCH=ON \ + -DBOOST_COROSIO_BUILD_TESTS=OFF \ + -DBOOST_COROSIO_BUILD_EXAMPLES=OFF \ + -DBOOST_COROSIO_BENCH_ASIO_IO_URING=OFF + cmake --build build-aoff --config Release --target corosio_bench --parallel + + # Linux only: a second binary with asio built against io_uring, so + # the asio-uring/asio_callback-uring configs compare against a + # same-reactor asio rather than an epoll-only one. Harmless to skip + # on Windows/macOS, which have no io_uring reactor to build against. + - name: Configure and build (io_uring asio reactor) + if: matrix.platform == 'linux' + run: | + set -e + cmake -S boost-root -B build-aon \ + -DCMAKE_BUILD_TYPE=Release \ + -DBOOST_INCLUDE_LIBRARIES="corosio;asio" \ + -DBOOST_COROSIO_BUILD_BENCH=ON \ + -DBOOST_COROSIO_BUILD_TESTS=OFF \ + -DBOOST_COROSIO_BUILD_EXAMPLES=OFF \ + -DBOOST_COROSIO_BENCH_ASIO_IO_URING=ON + cmake --build build-aon --config Release --target corosio_bench --parallel + + - name: Locate binaries + id: bin + run: | + set -e + bin_off=$(find build-aoff -type f \( -name corosio_bench -o -name corosio_bench.exe \) | head -1) + test -n "$bin_off" + "$bin_off" --library asio --list > /dev/null # hard-fail early if asio absent + if [ "${{ matrix.platform }}" = "linux" ]; then + bin_on=$(find build-aon -type f -name corosio_bench | head -1) + test -n "$bin_on" + "$bin_on" --library asio --list > /dev/null + else + bin_on="$bin_off" + fi + echo "bin_off=$bin_off" >> "$GITHUB_OUTPUT" + echo "bin_on=$bin_on" >> "$GITHUB_OUTPUT" + + - name: Run interleaved comparison + run: | + set -e + configs=(${{ matrix.configs }}) + n=${#configs[@]} + for i in $(seq 1 "$ITERS"); do + # rotate starting configuration each iteration so slow drift + # does not systematically favor any one implementation + for j in $(seq 0 $((n - 1))); do + cfg=${configs[$(( (j + i - 1) % n ))]} + case "$cfg" in + corosio-*) args="--library corosio --backend ${cfg#corosio-}" ;; + asio-*) args="--library asio --backend ${cfg#asio-}" ;; + asio_callback-*) args="--library asio_callback --backend ${cfg#asio_callback-}" ;; + *) args="--library $cfg" ;; + esac + # Only the io_uring-flavored asio configs need the ON binary + # (asio's reactor is fixed at compile time); corosio picks + # its backend at runtime via --backend, so every corosio + # config runs fine on the OFF binary. + case "$cfg" in + asio-uring|asio_callback-uring) bin="${{ steps.bin.outputs.bin_on }}" ;; + *) bin="${{ steps.bin.outputs.bin_off }}" ;; + esac + "$bin" $args \ + --duration "$DUR" --output "results/$cfg-$i.json" + done + done + + - name: Capture environment + continue-on-error: true + run: | + cpu=$( (grep -m1 'model name' /proc/cpuinfo | cut -d: -f2-) 2>/dev/null \ + || sysctl -n machdep.cpu.brand_string 2>/dev/null \ + || powershell -NoProfile -Command "(Get-CimInstance Win32_Processor).Name" 2>/dev/null \ + || echo unknown) + cores=$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo unknown) + ram=$( (awk '/MemTotal/ {printf "%.0f", $2/1048576}' /proc/meminfo) 2>/dev/null \ + || (sysctl -n hw.memsize 2>/dev/null | awk '{printf "%.0f", $1/1073741824}') \ + || powershell -NoProfile -Command "[math]::Round((Get-CimInstance Win32_ComputerSystem).TotalPhysicalMemory/1GB)" 2>/dev/null \ + || echo unknown) + osname=$(uname -sr 2>/dev/null || echo unknown) + kern=$(uname -v 2>/dev/null || echo unknown) + if command -v cl.exe >/dev/null 2>&1; then comp=$(cl.exe 2>&1 | head -1) + elif command -v c++ >/dev/null 2>&1; then comp=$(c++ --version | head -1) + else comp=unknown; fi + cmakev=$(cmake --version | head -1) + uringv=$( (dpkg-query -W -f '${Version}' liburing-dev) 2>/dev/null || echo n/a) + ASIO_REACTOR=$(case "${{ matrix.platform }}" in linux) echo "epoll and io_uring (matched per configuration)";; windows) echo IOCP;; macos) echo kqueue;; esac) + CPU="$cpu" CORES="$cores" RAM="$ram" OSN="$osname" KERN="$kern" \ + COMP="$comp" CMAKEV="$cmakev" URINGV="$uringv" \ + BOOST_SHA=$(git -C boost-root rev-parse HEAD) \ + ASIO_SHA=$(git -C boost-root/libs/asio rev-parse HEAD 2>/dev/null || echo unknown) \ + CAPY_SHA=$(git -C ws/capy rev-parse HEAD) \ + COROSIO_SHA=${{ github.sha }} COROSIO_REF=${{ github.ref_name }} \ + PLATFORM=${{ matrix.platform }} ITERS="$ITERS" DUR="$DUR" \ + ASIO_REACTOR="$ASIO_REACTOR" \ + RUN_URL="${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" \ + python3 - <<'EOF' + import json, os + e = os.environ + json.dump({ + "platform": e["PLATFORM"], "cpu": e["CPU"].strip(), + "cores": e["CORES"], "ram_gb": e["RAM"], "os": e["OSN"], + "kernel": e["KERN"], "compiler": e["COMP"], "cmake": e["CMAKEV"], + "liburing": e["URINGV"], "boost_sha": e["BOOST_SHA"], + "asio_sha": e["ASIO_SHA"], "asio_reactor": e["ASIO_REACTOR"], + "capy_sha": e["CAPY_SHA"], + "corosio_sha": e["COROSIO_SHA"], "corosio_ref": e["COROSIO_REF"], + "date_utc": __import__("datetime").datetime.utcnow().strftime("%Y-%m-%d"), + "iterations": int(e["ITERS"]), "duration_s": float(e["DUR"]), + "run_url": e["RUN_URL"], + }, open("results/environment.json", "w"), indent=2) + EOF + + - name: Upload results + if: always() + uses: actions/upload-artifact@v4 + with: + name: bench-report-${{ matrix.platform }} + path: results/ diff --git a/bench/README.md b/bench/README.md new file mode 100644 index 000000000..04db04b3e --- /dev/null +++ b/bench/README.md @@ -0,0 +1,150 @@ +# Benchmarks + +## What lives here + +- `corosio/` — corosio benchmark suites (one `.cpp` per category). +- `asio/callback/`, `asio/coroutine/` — equivalent suites against + Boost.Asio, built only when `Boost::asio` is available at configure time. +- `common/` — shared harness: benchmark registration, timing, backend + selection, HTTP parsing helpers. + +All suites build into a single `corosio_bench` binary, selected at +runtime via `--library`. Categories: `io_context`, `socket_throughput`, +`socket_latency`, `http_server`, `accept_churn`, `fan_out`, +`local_socket_throughput`, `local_socket_latency` (the last two are POSIX +only). Run `corosio_bench --list` for the exact set of benchmarks in each +category on your build — some are gated by backend or platform. + +## Building + +Build inside the Boost superproject with `asio` included, so +`bench/CMakeLists.txt` picks up `Boost::asio` as a sibling target and +compiles the comparison suites: + +```bash +cmake -S boost-root -B build \ + -DCMAKE_BUILD_TYPE=Release \ + -DBOOST_INCLUDE_LIBRARIES="corosio;asio" \ + -DBOOST_COROSIO_BUILD_BENCH=ON +cmake --build build --config Release --target corosio_bench --parallel +``` + +Building corosio standalone also works +(`-DBOOST_COROSIO_BUILD_BENCH=ON` from the corosio checkout), but then +the comparison suites need a system-installed Boost 1.84+ with Asio — +`bench/CMakeLists.txt` falls back to `find_package(Boost COMPONENTS +asio)` when no sibling target exists. Either way, if configure prints: + +``` +Boost.Asio not found -- comparison benchmarks disabled +``` + +then `corosio_bench` only accepts `--library corosio`; `asio` and +`asio_callback` are compiled out entirely, not just unavailable at +runtime. + +Always build `Release`. Debug disables inlining and optimization, which +distorts relative costs between fast and slow paths. + +## Running locally + +The binary lands at `build/bench/<config>/corosio_bench` (or +`build/bench/corosio_bench` for a single-config generator). Run +benchmarks one process at a time — never in parallel — since concurrent +processes contend for CPU and cache and produce unreliable numbers. + +Full suite, default library (corosio), platform-default backend: + +```bash +build/bench/Release/corosio_bench +``` + +One category: + +```bash +build/bench/Release/corosio_bench --category socket_throughput +``` + +One benchmark (`--bench` is a prefix match on the benchmark name, not +the category; combine with `--category` to disambiguate across +categories): + +```bash +build/bench/Release/corosio_bench --category socket_throughput \ + --bench unidirectional +``` + +Select a backend, comparison library, duration, and warmup: + +```bash +build/bench/Release/corosio_bench --library asio_callback --backend epoll \ + --duration 5 --warmup 0.5 +``` + +`--duration` sets the measured time per benchmark (default 3s); +`--warmup` runs an unmeasured pass first (default 0, disabled) — use it +for benchmarks sensitive to cold caches or lazy connection setup. +`--library all` runs corosio and both asio variants back to back. Write +results to JSON for later aggregation: + +```bash +build/bench/Release/corosio_bench --output results/corosio-epoll-1.json +``` + +To compare two working trees (e.g. before/after a change), build both, +then run each several times with the same flags, alternating which one +goes first each iteration — this cancels out drift from thermal +throttling or background load instead of it favoring whichever side ran +first. + +## The PR benchmark workflow + +`.github/workflows/benchmarks.yml` runs `corosio_bench` on dedicated +self-hosted runners for pull requests opened by the repo owner or a +collaborator, or on any PR labeled `benchmark`. It builds base and head, +runs both interleaved (ABBA order) across several iterations per +platform, and posts a single updating PR comment +(`.github/bench/compare.py`) summarizing per-category deltas with a +within-noise/faster/slower verdict. It's advisory only — see issue #343. + +## Refreshing the published report + +`doc/modules/ROOT/pages/benchmark-report.adoc` is a generated page — +nothing on it is hand-edited. Refresh it after a change that meaningfully +affects performance: + +```bash +# 1. Run the full suite on all three platforms and collect raw JSON. +gh workflow run benchmark-report.yml --repo cppalliance/corosio --ref develop + +# 2. Download the workflow's artifacts once it completes. +gh run download <run-id> --repo cppalliance/corosio -D /tmp/bench-report + +# 3. Regenerate the page and its charts from the raw JSON. +python3 .github/bench/report_page.py --input-dir /tmp/bench-report \ + --output-dir doc/modules/ROOT + +# 4. Preview the rendered docs site. +./doc/build_antora.sh + +# 5. Commit the regenerated page and charts (nothing else). +git add doc/modules/ROOT/pages/benchmark-report.adoc doc/modules/ROOT/images/bench +git commit -m "docs: regenerate benchmark report" +``` + +Never regenerate the published page from a reduced/smoke run (e.g. +`iterations=1`): single-iteration data has no noise estimate, so the +summary's within-noise/faster/slower classifications are meaningless. + +`report_page.py` expects `<input-dir>` to contain one +`bench-report-<platform>/` directory per platform (`linux`, `windows`, +`macos`), each holding the raw `<config>-<iter>.json` files plus an +`environment.json`. A missing platform directory is not an error — the +page notes it and generation continues with what's present. + +## Methodology + +Metric selection, aggregation (median across iterations, CV for noise), +and the within-noise threshold are implemented once, in +`report_page.py`, and described in full on the generated page's own +Methodology section — that page is the source of truth, not this file. diff --git a/bench/common/benchmark.hpp b/bench/common/benchmark.hpp index f3eb199e8..b349eeca8 100644 --- a/bench/common/benchmark.hpp +++ b/bench/common/benchmark.hpp @@ -16,6 +16,7 @@ #include <ctime> #include <fstream> #include <iomanip> +#include <map> #include <sstream> #include <string> #include <vector> @@ -37,6 +38,7 @@ struct benchmark_result std::string library; std::string category; std::string name; + std::string description; std::vector<metric> metrics; benchmark_result(std::string lib, std::string cat, std::string n) @@ -74,6 +76,7 @@ class result_collector std::string timestamp_; double duration_s_ = 0.0; std::vector<benchmark_result> results_; + std::map<std::string, std::string> category_descriptions_; static std::string escape_json(std::string const& s) { @@ -147,6 +150,12 @@ class result_collector results_.push_back(std::move(result)); } + /// Record the description for a benchmark category. + void set_category_description(std::string category, std::string text) + { + category_descriptions_[std::move(category)] = std::move(text); + } + /** Serialize all results to JSON. */ std::string to_json() const { @@ -159,6 +168,22 @@ class result_collector oss << " \"timestamp\": \"" << escape_json(timestamp_) << "\",\n"; oss << " \"duration_s\": " << duration_s_ << "\n"; oss << " },\n"; + + if (!category_descriptions_.empty()) + { + oss << " \"category_descriptions\": {\n"; + std::size_t i = 0; + for (auto const& [category, text] : category_descriptions_) + { + oss << " \"" << escape_json(category) + << "\": \"" << escape_json(text) << "\""; + if (++i < category_descriptions_.size()) + oss << ","; + oss << "\n"; + } + oss << " },\n"; + } + oss << " \"benchmarks\": [\n"; for (std::size_t i = 0; i < results_.size(); ++i) @@ -172,6 +197,10 @@ class result_collector << "\",\n"; oss << " \"name\": \"" << escape_json(r.name) << "\""; + if (!r.description.empty()) + oss << ",\n \"description\": \"" + << escape_json(r.description) << "\""; + for (auto const& m : r.metrics) oss << ",\n \"" << escape_json(m.name) << "\": " << m.value; diff --git a/bench/common/suite.hpp b/bench/common/suite.hpp index d7f0bec51..9f373b69e 100644 --- a/bench/common/suite.hpp +++ b/bench/common/suite.hpp @@ -262,6 +262,7 @@ struct suite_entry std::function<void(state&)> fn; bench_flags flags = bench_flags::none; std::vector<int64_t> args; + std::string description; }; /** Group of related benchmarks sharing a category name. */ @@ -269,6 +270,7 @@ class benchmark_suite { std::string library_; std::string category_; + std::string category_description_; bench_flags flags_; std::vector<suite_entry> entries_; @@ -286,7 +288,7 @@ class benchmark_suite benchmark_suite& add(std::string name, bench_fn fn, bench_flags flags = bench_flags::none) { - entries_.push_back({std::move(name), std::move(fn), flags, {}}); + entries_.push_back({std::move(name), std::move(fn), flags, {}, {}}); return *this; } @@ -298,6 +300,14 @@ class benchmark_suite return *this; } + /// Set the description for the most recently added benchmark. + benchmark_suite& describe(std::string text) + { + if (!entries_.empty()) + entries_.back().description = std::move(text); + return *this; + } + /// Generate a range of argument values for the most recently /// added benchmark: lo, lo*mul, lo*mul*mul, ... up to hi. benchmark_suite& range(int64_t lo, int64_t hi, int64_t mul) @@ -316,6 +326,13 @@ class benchmark_suite library_ = std::move(lib); } + /// Set the description for this suite's category. + benchmark_suite& category_description(std::string text) + { + category_description_ = std::move(text); + return *this; + } + std::string const& library() const { return library_; @@ -324,6 +341,10 @@ class benchmark_suite { return category_; } + std::string const& category_description() const + { + return category_description_; + } bench_flags flags() const { return flags_; @@ -451,6 +472,10 @@ class benchmark_runner !explicit_cat && !enable_microbenchmarks) continue; + if (!suite.category_description().empty()) + collector_.set_category_description( + suite.category(), suite.category_description()); + for (auto const& entry : suite.entries()) { bool needs_drain = @@ -465,7 +490,7 @@ class benchmark_runner run_entry( suite.library(), suite.category(), entry.name, entry.fn, - {}, needs_drain); + entry.description, {}, needs_drain); } else { @@ -478,7 +503,7 @@ class benchmark_runner run_entry( suite.library(), suite.category(), full_name, - entry.fn, {v}, needs_drain); + entry.fn, entry.description, {v}, needs_drain); } } } @@ -497,6 +522,7 @@ class benchmark_runner std::string const& category, std::string const& name, std::function<void(state&)> const& fn, + std::string const& description, std::vector<int64_t> ranges, bool needs_drain) { @@ -529,7 +555,7 @@ class benchmark_runner fn(st); print_results(st); - collect_results(library, category, name, st); + collect_results(library, category, name, description, st); } void print_results(state const& st) @@ -611,6 +637,7 @@ class benchmark_runner std::string const& library, std::string const& category, std::string const& name, + std::string const& description, state const& st) { double elapsed = st.elapsed_seconds(); @@ -618,6 +645,7 @@ class benchmark_runner return; benchmark_result result(library, category, name); + result.description = description; result.add("elapsed_s", elapsed); int64_t ops = st.total_ops(); diff --git a/bench/corosio/accept_churn_bench.cpp b/bench/corosio/accept_churn_bench.cpp index 858d94b63..8bab116d8 100644 --- a/bench/corosio/accept_churn_bench.cpp +++ b/bench/corosio/accept_churn_bench.cpp @@ -497,13 +497,33 @@ make_accept_churn_suite() { using F = bench::bench_flags; return bench::benchmark_suite("accept_churn", F::needs_conntrack_drain) + .category_description( + "Rate of setting up and tearing down short-lived TCP " + "connections: connect, accept, close.") .add("sequential", bench_sequential_churn<Backend>) + .describe( + "Single connect/accept/close loop with a one-way one-byte " + "transfer (client writes, server reads) per connection, one " + "connection at a time.") .add("sequential_lockless", bench_sequential_churn_lockless<Backend>) + .describe( + "Same as sequential, with the context in single-threaded " + "lockless mode.") .add("concurrent", bench_concurrent_churn<Backend>) + .describe( + "N independent accept loops run on separate listeners at once, " + "each doing a one-way one-byte transfer (client writes, server " + "reads) per connection; N is the number of loops.") .args({1, 4, 16}) .add("burst", bench_burst_churn<Backend>) + .describe( + "N connects are fired at once, then all N are accepted before " + "closing; N is the burst size.") .args({10, 100}) .add("burst_lockless", bench_burst_churn_lockless<Backend>) + .describe( + "Same as burst, with the context in single-threaded lockless " + "mode; N is the burst size.") .args({10, 100}); } diff --git a/bench/corosio/fan_out_bench.cpp b/bench/corosio/fan_out_bench.cpp index cf573d10b..fcaf21df0 100644 --- a/bench/corosio/fan_out_bench.cpp +++ b/bench/corosio/fan_out_bench.cpp @@ -550,19 +550,42 @@ make_fan_out_suite() { using F = bench::bench_flags; return bench::benchmark_suite("fan_out", F::needs_conntrack_drain) + .category_description( + "Fan-out/fan-in coroutine coordination: a parent starts " + "concurrent sub-requests against echo servers and awaits their " + "completion via a shared latch.") .add("fork_join", bench_fork_join<Backend>) + .describe( + "One parent starts N sub-requests to echo servers and waits " + "for all to finish before repeating; N is the fan-out width.") .args({1, 4, 16, 64}) .add("fork_join_lockless", bench_fork_join_lockless<Backend>) + .describe( + "Same as fork_join, with the context in single-threaded " + "lockless mode; N is the fan-out width.") .args({1, 4, 16, 64}) .add("nested", bench_nested<Backend>) + .describe( + "Two-level fan-out: the parent starts N groups of 4 " + "sub-requests each, each group awaited via its own latch; N " + "is the number of groups.") .args({4, 16}) .add("nested_lockless", bench_nested_lockless<Backend>) + .describe( + "Same as nested, with the context in single-threaded lockless " + "mode; N is the number of groups.") .args({4, 16}) .add("concurrent_parents", bench_concurrent_parents<Backend>) + .describe( + "N independent parents each fan out to 16 sub-requests at " + "once; N is the number of parents.") .args({1, 4, 16}) .add( "concurrent_parents_lockless", bench_concurrent_parents_lockless<Backend>) + .describe( + "Same as concurrent_parents, with the context in " + "single-threaded lockless mode; N is the number of parents.") .args({1, 4, 16}); } diff --git a/bench/corosio/http_server_bench.cpp b/bench/corosio/http_server_bench.cpp index 5fd7669e8..4ef75cf58 100644 --- a/bench/corosio/http_server_bench.cpp +++ b/bench/corosio/http_server_bench.cpp @@ -329,11 +329,27 @@ make_http_server_suite() using F = bench::bench_flags; return bench::benchmark_suite("http_server", F::needs_conntrack_drain) + .category_description( + "Request/response throughput of a minimal HTTP/1.1 server " + "exchanging a fixed small request and canned response over " + "persistent TCP loopback connections.") .add("single_conn", bench_single_connection<Backend>) + .describe( + "One client repeatedly sends a fixed small HTTP request to " + "one server and reads the response.") .add("single_conn_lockless", bench_single_connection_lockless<Backend>) + .describe( + "Same as single_conn, with the context in single-threaded " + "lockless mode.") .add("concurrent", bench_concurrent_connections<Backend>) + .describe( + "N client/server pairs run the single_conn request/response " + "loop concurrently; N is the number of connections.") .args({1, 4, 16, 32}) .add("multithread", bench_multithread<Backend>) + .describe( + "32 client/server pairs share one context serviced by N " + "threads running the context; N is the thread count.") .args({1, 2, 4, 8, 16}); } diff --git a/bench/corosio/io_context_bench.cpp b/bench/corosio/io_context_bench.cpp index be38f756b..d8a299754 100644 --- a/bench/corosio/io_context_bench.cpp +++ b/bench/corosio/io_context_bench.cpp @@ -330,17 +330,47 @@ make_io_context_suite() { using F = bench::bench_flags; return bench::benchmark_suite("io_context", F::is_microbenchmark) + .category_description( + "Overhead of posting and dispatching handlers on the " + "io_context scheduler itself, with no actual I/O involved.") .add("single_threaded", bench_single_threaded_post<Backend>) + .describe( + "One thread repeatedly posts batches of 1000 minimal " + "counter-increment handlers, then drains them with " + "poll()/restart().") .add("multithreaded", bench_multithreaded_scaling<Backend>) + .describe( + "One feeder thread continuously posts batches of 100000 " + "atomic-increment handlers while N worker threads poll and " + "restart the shared context; N is the worker thread count.") .args({8}) .add("interleaved", bench_interleaved_post_run<Backend>) + .describe( + "Same as single_threaded, but with smaller batches of 100 " + "handlers per post/poll/restart cycle.") .add("concurrent", bench_concurrent_post_run<Backend>) + .describe( + "N threads each post their own batches of 10000 " + "atomic-increment handlers and poll/restart the shared " + "context themselves; N is the thread count.") .args({4}) .add("high_inline_budget", bench_high_inline_budget<Backend>) + .describe( + "Same as single_threaded, but the context may run up to 64 " + "handlers inline per dispatch instead of the default 16.") .add("large_event_buffer", bench_large_event_buffer<Backend>) + .describe( + "Same as single_threaded, but the context polls up to 512 " + "events per iteration instead of the default 128.") .add( "single_threaded_lockless", bench_single_threaded_lockless<Backend>) - .add("interleaved_lockless", bench_interleaved_lockless<Backend>); + .describe( + "Same as single_threaded, with the context in single-threaded " + "lockless mode.") + .add("interleaved_lockless", bench_interleaved_lockless<Backend>) + .describe( + "Same as interleaved, with the context in single-threaded " + "lockless mode."); } } // namespace corosio_bench diff --git a/bench/corosio/local_socket_latency_bench.cpp b/bench/corosio/local_socket_latency_bench.cpp index 159ea67c9..77403732a 100644 --- a/bench/corosio/local_socket_latency_bench.cpp +++ b/bench/corosio/local_socket_latency_bench.cpp @@ -253,15 +253,31 @@ make_local_socket_latency_suite() using F = bench::bench_flags; return bench::benchmark_suite("local_socket_latency", F::none) + .category_description( + "Round-trip latency of a single Unix domain stream socket " + "write/read exchange across message sizes and concurrent " + "pair counts.") .add("pingpong", bench_unix_pingpong_latency<Backend>) + .describe( + "One connected Unix domain socket pair ping-pongs a message " + "client->server->client; N is the message size in bytes.") .args({1, 64, 1024}) .add("pingpong_lockless", bench_unix_pingpong_latency_lockless<Backend>) + .describe( + "Same as pingpong, with the context in single-threaded lockless " + "mode; N is the message size in bytes.") .args({1, 64, 1024}) .add("concurrent", bench_unix_concurrent_latency<Backend>) + .describe( + "N independent 64-byte pingpong pairs run concurrently on one " + "context; N is the number of connection pairs.") .args({1, 4, 16}) .add( "concurrent_lockless", bench_unix_concurrent_latency_lockless<Backend>) + .describe( + "Same as concurrent, with the context in single-threaded " + "lockless mode; N is the number of connection pairs.") .args({1, 4, 16}); } diff --git a/bench/corosio/local_socket_throughput_bench.cpp b/bench/corosio/local_socket_throughput_bench.cpp index 1aad2b72f..41a11e8dc 100644 --- a/bench/corosio/local_socket_throughput_bench.cpp +++ b/bench/corosio/local_socket_throughput_bench.cpp @@ -356,15 +356,30 @@ make_local_socket_throughput_suite() using F = bench::bench_flags; return bench::benchmark_suite("local_socket_throughput", F::none) + .category_description( + "Sustained byte throughput of a Unix domain stream socket pair " + "under continuous streaming across chunk sizes and directions.") .add("unidirectional", bench_unix_throughput<Backend>) + .describe( + "One writer streams to one reader as fast as possible; N is " + "the write/read chunk size in bytes.") .range(1024, 1048576, 4) .add("unidirectional_lockless", bench_unix_throughput_lockless<Backend>) + .describe( + "Same as unidirectional, with the context in single-threaded " + "lockless mode; N is the chunk size in bytes.") .range(1024, 1048576, 4) .add("bidirectional", bench_unix_bidirectional_throughput<Backend>) + .describe( + "Both ends of the socket pair write and read simultaneously; N " + "is the chunk size in bytes.") .range(1024, 1048576, 4) .add( "bidirectional_lockless", bench_unix_bidirectional_throughput_lockless<Backend>) + .describe( + "Same as bidirectional, with the context in single-threaded " + "lockless mode; N is the chunk size in bytes.") .range(1024, 1048576, 4); } diff --git a/bench/corosio/socket_latency_bench.cpp b/bench/corosio/socket_latency_bench.cpp index bc3eeb4bf..4aff46b93 100644 --- a/bench/corosio/socket_latency_bench.cpp +++ b/bench/corosio/socket_latency_bench.cpp @@ -256,13 +256,29 @@ make_socket_latency_suite() using F = bench::bench_flags; return bench::benchmark_suite("socket_latency", F::needs_conntrack_drain) + .category_description( + "Round-trip latency of a TCP loopback connection. Each sample " + "is one full round trip (request out, reply back) across " + "message sizes and concurrent pair counts.") .add("pingpong", bench_pingpong_latency<Backend>) + .describe( + "One TCP connection ping-pongs a message client->server->client; " + "N is the message size in bytes.") .args({1, 64, 1024}) .add("pingpong_lockless", bench_pingpong_latency_lockless<Backend>) + .describe( + "Same as pingpong, with the context in single-threaded lockless " + "mode; N is the message size in bytes.") .args({1, 64, 1024}) .add("concurrent", bench_concurrent_latency<Backend>) + .describe( + "N independent 64-byte pingpong pairs run concurrently on one " + "context; N is the number of connection pairs.") .args({1, 4, 16}) .add("concurrent_lockless", bench_concurrent_latency_lockless<Backend>) + .describe( + "Same as concurrent, with the context in single-threaded " + "lockless mode; N is the number of connection pairs.") .args({1, 4, 16}); } diff --git a/bench/corosio/socket_throughput_bench.cpp b/bench/corosio/socket_throughput_bench.cpp index 43345f4b2..87188d24d 100644 --- a/bench/corosio/socket_throughput_bench.cpp +++ b/bench/corosio/socket_throughput_bench.cpp @@ -497,17 +497,37 @@ make_socket_throughput_suite() using F = bench::bench_flags; return bench::benchmark_suite("socket_throughput", F::needs_conntrack_drain) + .category_description( + "Sustained byte throughput of a TCP loopback connection under " + "continuous streaming, varying chunk size, direction, and " + "concurrency.") .add("unidirectional", bench_throughput<Backend>) + .describe( + "One writer streams to one reader as fast as possible; N is " + "the write/read chunk size in bytes.") .range(1024, 1048576, 4) .add("unidirectional_lockless", bench_throughput_lockless<Backend>) + .describe( + "Same as unidirectional, with the context in single-threaded " + "lockless mode; N is the chunk size in bytes.") .range(1024, 1048576, 4) .add("bidirectional", bench_bidirectional_throughput<Backend>) + .describe( + "Both ends of the connection write and read simultaneously; N " + "is the chunk size in bytes.") .range(1024, 1048576, 4) .add( "bidirectional_lockless", bench_bidirectional_throughput_lockless<Backend>) + .describe( + "Same as bidirectional, with the context in single-threaded " + "lockless mode; N is the chunk size in bytes.") .range(1024, 1048576, 4) .add("multithread", bench_multithread_throughput<Backend>) + .describe( + "32 bidirectional connection pairs share one context serviced " + "by N threads running the context, with 64 KiB chunks; N is " + "the thread count.") .args({2, 4, 8}); } diff --git a/doc/.vale/styles/config/vocabularies/Corosio/accept.txt b/doc/.vale/styles/config/vocabularies/Corosio/accept.txt index 3275f5036..b308e4c5c 100644 --- a/doc/.vale/styles/config/vocabularies/Corosio/accept.txt +++ b/doc/.vale/styles/config/vocabularies/Corosio/accept.txt @@ -281,3 +281,6 @@ (?i)IP[’']s (?i)UDP[’']s (?i)URL[’']s +CMake +Ryzen +liburing diff --git a/doc/lint/baseline.json b/doc/lint/baseline.json index b6765ce81..d96e3614b 100644 --- a/doc/lint/baseline.json +++ b/doc/lint/baseline.json @@ -3,7 +3,7 @@ "note": "Snapshot of current violations (Style Guide Part F.0). Everything recorded here is grandfathered; check-no-new-violations.mjs fails on findings that are NOT in it, for the rules named by the workflow's --gate spec. Reseed only via the workflow_dispatch steps, never locally. Fingerprints are line-insensitive: the `#N` component is the Nth occurrence of that (file, message) pair, NOT a line number, so inserting text above a finding does not rename it.", "checks": { "vale_adoc": { - "count": 55, + "count": 45, "skipped": false, "fingerprints": [ "modules/ROOT/pages/2.networking-tutorial/2b.internet-addresses.adoc:#1:Google.OxfordComma", @@ -48,16 +48,6 @@ "modules/ROOT/pages/4.guide/4p.unix-sockets.adoc:#1:Google.OxfordComma", "modules/ROOT/pages/4.guide/4q.udp.adoc:#1:Google.Colons", "modules/ROOT/pages/4.guide/4q.udp.adoc:#1:Google.OxfordComma", - "modules/ROOT/pages/benchmark-report.adoc:#1:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#1:Google.Units", - "modules/ROOT/pages/benchmark-report.adoc:#2:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#2:Google.Units", - "modules/ROOT/pages/benchmark-report.adoc:#3:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#4:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#5:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#6:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#7:Google.Colons", - "modules/ROOT/pages/benchmark-report.adoc:#8:Google.Colons", "modules/ROOT/pages/glossary.adoc:#1:Google.OxfordComma", "modules/ROOT/pages/glossary.adoc:#1:Vale.Spelling", "modules/ROOT/pages/quick-start.adoc:#1:Vale.Spelling" diff --git a/doc/modules/ROOT/images/bench/linux-accept_churn-epoll.svg b/doc/modules/ROOT/images/bench/linux-accept_churn-epoll.svg new file mode 100644 index 000000000..bf72617b0 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-accept_churn-epoll.svg @@ -0,0 +1,51 @@ +<svg xmlns='http://www.w3.org/2000/svg' class='bch-chart' width='720' height='371' viewBox='0 0 720 371'> +<title>accept_churn (epoll) — Linux + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-15% + +-10% + +-5% + ++5% +0% +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +-4% +burst/10 +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +-13% +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/100 +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/100 +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/1 +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +-2% +concurrent/16 +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +-4% +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +sequential +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +sequential_lockless + + diff --git a/doc/modules/ROOT/images/bench/linux-accept_churn-uring.svg b/doc/modules/ROOT/images/bench/linux-accept_churn-uring.svg new file mode 100644 index 000000000..9b8e214d2 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-accept_churn-uring.svg @@ -0,0 +1,51 @@ + +accept_churn (io_uring) — Linux + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-60% + +-40% + +-20% + ++20% +0% +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/100 +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. ++0% +burst_lockless/100 +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +-3% +concurrent/1 +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +-35% +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/16 +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +sequential +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +-22% +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +sequential_lockless + + diff --git a/doc/modules/ROOT/images/bench/linux-fan_out-epoll-g1.svg b/doc/modules/ROOT/images/bench/linux-fan_out-epoll-g1.svg new file mode 100644 index 000000000..469552c13 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-fan_out-epoll-g1.svg @@ -0,0 +1,50 @@ + +fan_out (epoll) — Linux, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-15% + +-10% + +-5% + ++5% +0% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-8% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-5% +concurrent_parents/1 +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-9% +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-5% +concurrent_parents/16 +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-9% +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-5% +concurrent_parents/4 +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-8% +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-4% +concurrent_parents_lockless/1 +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-9% +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-5% +concurrent_parents_lockless/16 +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-9% +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-5% +concurrent_parents_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/linux-fan_out-epoll-g2.svg b/doc/modules/ROOT/images/bench/linux-fan_out-epoll-g2.svg new file mode 100644 index 000000000..3f597dc84 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-fan_out-epoll-g2.svg @@ -0,0 +1,60 @@ + +fan_out (epoll) — Linux, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-15% + +-10% + +-5% + ++5% +0% +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +-12% +fork_join/1 +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +-8% +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/4 +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/1 +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +-4% +fork_join_lockless/64 +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +-9% +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/16 +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/linux-fan_out-uring-g1.svg b/doc/modules/ROOT/images/bench/linux-fan_out-uring-g1.svg new file mode 100644 index 000000000..b1972783b --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-fan_out-uring-g1.svg @@ -0,0 +1,52 @@ + +fan_out (io_uring) — Linux, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-20% + +-15% + +-10% + +-5% + ++5% +0% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-13% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-4% +concurrent_parents/1 +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-13% +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-3% +concurrent_parents/16 +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-13% +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-3% +concurrent_parents/4 +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-8% +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-4% +concurrent_parents_lockless/1 +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-9% +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-3% +concurrent_parents_lockless/16 +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-9% +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-3% +concurrent_parents_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/linux-fan_out-uring-g2.svg b/doc/modules/ROOT/images/bench/linux-fan_out-uring-g2.svg new file mode 100644 index 000000000..672e021ce --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-fan_out-uring-g2.svg @@ -0,0 +1,58 @@ + +fan_out (io_uring) — Linux, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-20% + +-10% + ++10% +0% +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +-12% +fork_join/1 +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/4 +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. ++2% +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/1 +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +-2% +fork_join_lockless/64 +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +-16% +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/16 +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/linux-http_server-epoll.svg b/doc/modules/ROOT/images/bench/linux-http_server-epoll.svg new file mode 100644 index 000000000..aea3b9430 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-http_server-epoll.svg @@ -0,0 +1,59 @@ + +http_server (epoll) — Linux + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-10% + +-7.5% + +-5% + +-2.5% + ++2.5% +0% +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +-1% +concurrent/4 +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +-1% +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/16 +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +-8% +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +-5% +multithread/2 +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +single_conn_lockless + + diff --git a/doc/modules/ROOT/images/bench/linux-http_server-uring.svg b/doc/modules/ROOT/images/bench/linux-http_server-uring.svg new file mode 100644 index 000000000..637b9de0e --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-http_server-uring.svg @@ -0,0 +1,57 @@ + +http_server (io_uring) — Linux + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-50% + ++50% + ++100% + ++150% +0% +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +-14% +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +-7% +multithread/16 +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. ++91% +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +0% +multithread/2 +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +single_conn_lockless + + diff --git a/doc/modules/ROOT/images/bench/linux-local_socket_latency-epoll.svg b/doc/modules/ROOT/images/bench/linux-local_socket_latency-epoll.svg new file mode 100644 index 000000000..559275a43 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-local_socket_latency-epoll.svg @@ -0,0 +1,62 @@ + +local_socket_latency (epoll) — Linux + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-20% + +-10% + ++10% + ++20% + ++30% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-8% +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-8% +concurrent_lockless/4 +pingpong/1 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 +pingpong/1024 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. ++18% +pingpong/1024 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 +pingpong/64 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +-4% +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/linux-local_socket_latency-uring.svg b/doc/modules/ROOT/images/bench/linux-local_socket_latency-uring.svg new file mode 100644 index 000000000..3374638b0 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-local_socket_latency-uring.svg @@ -0,0 +1,58 @@ + +local_socket_latency (io_uring) — Linux + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-20% + ++20% + ++40% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-11% +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 +pingpong/1 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. ++26% +pingpong/1 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 +pingpong/1024 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +-8% +pingpong/1024 +pingpong/64 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. ++2% +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g1.svg b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g1.svg new file mode 100644 index 000000000..cf9432f23 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g1.svg @@ -0,0 +1,62 @@ + +local_socket_throughput (epoll) — Linux, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-30% + +-20% + +-10% + ++10% + ++20% +0% +bidirectional/1024 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +-11% +bidirectional/1024 +bidirectional/1048576 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 +bidirectional/262144 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 +bidirectional/4096 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. ++8% +bidirectional/4096 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++0% +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-19% +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g2.svg b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g2.svg new file mode 100644 index 000000000..db23ea332 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-epoll-g2.svg @@ -0,0 +1,58 @@ + +local_socket_throughput (epoll) — Linux, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-20% + ++20% + ++40% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. ++16% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +-6% +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-15% +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++1% +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g1.svg b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g1.svg new file mode 100644 index 000000000..1d5eb3dc9 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g1.svg @@ -0,0 +1,58 @@ + +local_socket_throughput (io_uring) — Linux, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-40% + +-20% + ++20% +0% +bidirectional/1024 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +-5% +bidirectional/1024 +bidirectional/1048576 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 +bidirectional/262144 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 +bidirectional/4096 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +-20% +bidirectional/65536 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++12% +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++3% +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g2.svg b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g2.svg new file mode 100644 index 000000000..e50aff8c2 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-local_socket_throughput-uring-g2.svg @@ -0,0 +1,60 @@ + +local_socket_throughput (io_uring) — Linux, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-40% + +-20% + ++20% + ++40% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +-9% +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++17% +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++3% +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-18% +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_latency-epoll.svg b/doc/modules/ROOT/images/bench/linux-socket_latency-epoll.svg new file mode 100644 index 000000000..b98400290 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_latency-epoll.svg @@ -0,0 +1,58 @@ + +socket_latency (epoll) — Linux + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-5% + ++5% + ++10% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-3% +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-3% +concurrent_lockless/4 +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. ++7% +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +-2% +pingpong/1024 +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_latency-uring.svg b/doc/modules/ROOT/images/bench/linux-socket_latency-uring.svg new file mode 100644 index 000000000..11a07cb9e --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_latency-uring.svg @@ -0,0 +1,58 @@ + +socket_latency (io_uring) — Linux + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-10% + ++10% + ++20% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-5% +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. ++13% +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +-2% +pingpong/1 +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +-3% +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g1.svg b/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g1.svg new file mode 100644 index 000000000..6d1a0c13b --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g1.svg @@ -0,0 +1,58 @@ + +socket_throughput (epoll) — Linux, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-10% + ++10% + ++20% +0% +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +-3% +bidirectional/1024 +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. ++13% +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. ++2% +bidirectional/262144 +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-5% +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g2.svg b/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g2.svg new file mode 100644 index 000000000..e49dadd23 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g2.svg @@ -0,0 +1,35 @@ + +socket_throughput (epoll) — Linux, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-20% + ++20% + ++40% + ++60% +0% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++45% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-3% +multithread/2 +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++38% +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++0% +multithread/4 +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++25% +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +0% +multithread/8 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g3.svg b/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g3.svg new file mode 100644 index 000000000..7648aecff --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_throughput-epoll-g3.svg @@ -0,0 +1,60 @@ + +socket_throughput (epoll) — Linux, part 3 + + +↑ faster · dotted line = asio (callbacks) · backend = epoll + +corosio + +asio (coroutines) + +-20% + ++20% + ++40% + ++60% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. ++35% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +-3% +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-6% +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++0% +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g1.svg b/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g1.svg new file mode 100644 index 000000000..7b827b4ab --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g1.svg @@ -0,0 +1,60 @@ + +socket_throughput (io_uring) — Linux, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-20% + +-10% + ++10% + ++20% +0% +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +-5% +bidirectional/16384 +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-8% +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++13% +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++1% +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g2.svg b/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g2.svg new file mode 100644 index 000000000..e348a6c0e --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g2.svg @@ -0,0 +1,35 @@ + +socket_throughput (io_uring) — Linux, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-100% + ++100% + ++200% + ++300% +0% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++123% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++7% +multithread/2 +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++206% +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-1% +multithread/4 +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++215% +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-1% +multithread/8 + + diff --git a/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g3.svg b/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g3.svg new file mode 100644 index 000000000..a1dae0e68 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-socket_throughput-uring-g3.svg @@ -0,0 +1,60 @@ + +socket_throughput (io_uring) — Linux, part 3 + + +↑ faster · dotted line = asio (callbacks) · backend = io_uring + +corosio + +asio (coroutines) + +-20% + ++20% + ++40% + ++60% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. ++51% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-2% +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-5% +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++2% +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/linux-summary-epoll.svg b/doc/modules/ROOT/images/bench/linux-summary-epoll.svg new file mode 100644 index 000000000..88bc3ba40 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-summary-epoll.svg @@ -0,0 +1,51 @@ + +Summary (epoll) — Linux + + +← slower · within noise · faster → per benchmark, vs Boost.Asio (callbacks) + +local_socket_latency +3 benchmarks slower than asio (callbacks) +3 +3 benchmarks within noise than asio (callbacks) +3 +6 benchmarks faster than asio (callbacks) +6 ++11.1% +socket_latency +1 benchmark slower than asio (callbacks) +3 benchmarks within noise than asio (callbacks) +3 +8 benchmarks faster than asio (callbacks) +8 ++5.8% +socket_throughput +3 benchmarks slower than asio (callbacks) +3 +11 benchmarks within noise than asio (callbacks) +11 +13 benchmarks faster than asio (callbacks) +13 ++2.4% +local_socket_throughput +4 benchmarks slower than asio (callbacks) +4 +14 benchmarks within noise than asio (callbacks) +14 +6 benchmarks faster than asio (callbacks) +6 +-0.8% +http_server +10 benchmarks slower than asio (callbacks) +10 +1 benchmark within noise than asio (callbacks) +-6.0% +fan_out +18 benchmarks slower than asio (callbacks) +18 +-8.5% +accept_churn +9 benchmarks slower than asio (callbacks) +9 +-9.3% + diff --git a/doc/modules/ROOT/images/bench/linux-summary-uring.svg b/doc/modules/ROOT/images/bench/linux-summary-uring.svg new file mode 100644 index 000000000..4c7ed13a7 --- /dev/null +++ b/doc/modules/ROOT/images/bench/linux-summary-uring.svg @@ -0,0 +1,56 @@ + +Summary (io_uring) — Linux + + +← slower · within noise · faster → per benchmark, vs Boost.Asio (callbacks) + +local_socket_latency +3 benchmarks slower than asio (callbacks) +3 +1 benchmark within noise than asio (callbacks) +8 benchmarks faster than asio (callbacks) +8 ++20.3% +socket_latency +2 benchmarks slower than asio (callbacks) +2 +2 benchmarks within noise than asio (callbacks) +2 +8 benchmarks faster than asio (callbacks) +8 ++10.7% +socket_throughput +3 benchmarks slower than asio (callbacks) +3 +8 benchmarks within noise than asio (callbacks) +8 +16 benchmarks faster than asio (callbacks) +16 ++5.1% +http_server +4 benchmarks slower than asio (callbacks) +4 +2 benchmarks within noise than asio (callbacks) +2 +5 benchmarks faster than asio (callbacks) +5 ++1.8% +local_socket_throughput +7 benchmarks slower than asio (callbacks) +7 +10 benchmarks within noise than asio (callbacks) +10 +7 benchmarks faster than asio (callbacks) +7 +-4.1% +fan_out +16 benchmarks slower than asio (callbacks) +16 +2 benchmarks within noise than asio (callbacks) +2 +-10.1% +accept_churn +9 benchmarks slower than asio (callbacks) +9 +-26.1% + diff --git a/doc/modules/ROOT/images/bench/macos-accept_churn.svg b/doc/modules/ROOT/images/bench/macos-accept_churn.svg new file mode 100644 index 000000000..d4b1b786b --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-accept_churn.svg @@ -0,0 +1,51 @@ + +accept_churn — macOS + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-60% + +-40% + +-20% + ++20% +0% +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +-41% +burst/100 +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +-28% +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/100 +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/1 +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +-1% +concurrent/16 +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +sequential +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. ++10% +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +sequential_lockless + + diff --git a/doc/modules/ROOT/images/bench/macos-fan_out-g1.svg b/doc/modules/ROOT/images/bench/macos-fan_out-g1.svg new file mode 100644 index 000000000..4a9ea2fa1 --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-fan_out-g1.svg @@ -0,0 +1,48 @@ + +fan_out — macOS, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-40% + +-20% + ++20% +0% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-5% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-11% +concurrent_parents/1 +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-7% +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-30% +concurrent_parents/16 +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-8% +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-5% +concurrent_parents/4 +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-5% +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-11% +concurrent_parents_lockless/1 +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-10% +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-31% +concurrent_parents_lockless/16 +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-11% +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-4% +concurrent_parents_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/macos-fan_out-g2.svg b/doc/modules/ROOT/images/bench/macos-fan_out-g2.svg new file mode 100644 index 000000000..c6a36f8fd --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-fan_out-g2.svg @@ -0,0 +1,62 @@ + +fan_out — macOS, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-80% + +-60% + +-40% + +-20% + ++20% +0% +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/1 +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +-4% +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/4 +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +-7% +fork_join/64 +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +-27% +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +-53% +fork_join_lockless/1 +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/64 +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/16 +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/macos-http_server.svg b/doc/modules/ROOT/images/bench/macos-http_server.svg new file mode 100644 index 000000000..056edc7eb --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-http_server.svg @@ -0,0 +1,55 @@ + +http_server — macOS + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-40% + +-20% + ++20% +0% +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +0% +concurrent/16 +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/16 +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/2 +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. ++12% +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +-6% +single_conn +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +-21% +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +single_conn_lockless + + diff --git a/doc/modules/ROOT/images/bench/macos-local_socket_latency.svg b/doc/modules/ROOT/images/bench/macos-local_socket_latency.svg new file mode 100644 index 000000000..4f284e0b6 --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-local_socket_latency.svg @@ -0,0 +1,60 @@ + +local_socket_latency — macOS + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-50% + ++50% + ++100% + ++150% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. ++0% +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. ++59% +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +-4% +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 +pingpong/1 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 +pingpong/1024 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 +pingpong/64 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One connected Unix domain socket pair ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. ++96% +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/macos-local_socket_throughput-g1.svg b/doc/modules/ROOT/images/bench/macos-local_socket_throughput-g1.svg new file mode 100644 index 000000000..c14652bb1 --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-local_socket_throughput-g1.svg @@ -0,0 +1,62 @@ + +local_socket_throughput — macOS, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-250% + ++250% + ++500% + ++750% + ++1000% +0% +bidirectional/1024 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 +bidirectional/1048576 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 +bidirectional/262144 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 +bidirectional/4096 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 — Both ends of the socket pair write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++709% +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +0% +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-3% +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++241% +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/macos-local_socket_throughput-g2.svg b/doc/modules/ROOT/images/bench/macos-local_socket_throughput-g2.svg new file mode 100644 index 000000000..9f8affb57 --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-local_socket_throughput-g2.svg @@ -0,0 +1,62 @@ + +local_socket_throughput — macOS, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-500% + ++500% + ++1000% + ++1500% + ++2000% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. ++1402% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +0% +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. ++466% +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-7% +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/macos-socket_latency.svg b/doc/modules/ROOT/images/bench/macos-socket_latency.svg new file mode 100644 index 000000000..e8137047e --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-socket_latency.svg @@ -0,0 +1,58 @@ + +socket_latency — macOS + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-50% + ++50% + ++100% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. ++8% +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. ++5% +pingpong/1024 +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. ++70% +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +-8% +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/macos-socket_throughput-g1.svg b/doc/modules/ROOT/images/bench/macos-socket_throughput-g1.svg new file mode 100644 index 000000000..bd16d73ea --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-socket_throughput-g1.svg @@ -0,0 +1,60 @@ + +socket_throughput — macOS, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-50% + ++50% + ++100% + ++150% +0% +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +-12% +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++104% +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-20% +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +0% +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/macos-socket_throughput-g2.svg b/doc/modules/ROOT/images/bench/macos-socket_throughput-g2.svg new file mode 100644 index 000000000..b62069b9c --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-socket_throughput-g2.svg @@ -0,0 +1,33 @@ + +socket_throughput — macOS, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-20% + +-10% + ++10% +0% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++2% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++2% +multithread/2 +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++1% +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++5% +multithread/4 +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-11% +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. ++0% +multithread/8 + + diff --git a/doc/modules/ROOT/images/bench/macos-socket_throughput-g3.svg b/doc/modules/ROOT/images/bench/macos-socket_throughput-g3.svg new file mode 100644 index 000000000..d572b230e --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-socket_throughput-g3.svg @@ -0,0 +1,60 @@ + +socket_throughput — macOS, part 3 + + +↑ faster · dotted line = asio (callbacks) · backend = kqueue + +corosio + +asio (coroutines) + +-50% + ++50% + ++100% + ++150% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +0% +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +-23% +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++125% +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++10% +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/macos-summary.svg b/doc/modules/ROOT/images/bench/macos-summary.svg new file mode 100644 index 000000000..81fe7912f --- /dev/null +++ b/doc/modules/ROOT/images/bench/macos-summary.svg @@ -0,0 +1,47 @@ + +Summary — macOS + + +← slower · within noise · faster → per benchmark, vs Boost.Asio (callbacks) + +local_socket_throughput +24 benchmarks faster than asio (callbacks) +24 ++467.2% +local_socket_latency +12 benchmarks faster than asio (callbacks) +12 ++95.7% +socket_latency +2 benchmarks within noise than asio (callbacks) +10 benchmarks faster than asio (callbacks) +10 ++66.3% +socket_throughput +3 benchmarks slower than asio (callbacks) +3 +7 benchmarks within noise than asio (callbacks) +7 +17 benchmarks faster than asio (callbacks) +17 ++63.3% +http_server +6 benchmarks slower than asio (callbacks) +6 +3 benchmarks within noise than asio (callbacks) +3 +2 benchmarks faster than asio (callbacks) +-4.5% +fan_out +16 benchmarks slower than asio (callbacks) +16 +2 benchmarks within noise than asio (callbacks) +-8.7% +accept_churn +5 benchmarks slower than asio (callbacks) +5 +1 benchmark within noise than asio (callbacks) +3 benchmarks faster than asio (callbacks) +3 +-10.2% + diff --git a/doc/modules/ROOT/images/bench/windows-accept_churn.svg b/doc/modules/ROOT/images/bench/windows-accept_churn.svg new file mode 100644 index 000000000..bd8ed3a41 --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-accept_churn.svg @@ -0,0 +1,49 @@ + +accept_churn — Windows + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-20% + +-10% + ++10% +0% +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/10 +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/100 — N connects are fired at once, then all N are accepted before closing; N is the burst size. +burst/100 +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +burst_lockless/10 +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. +-8% +burst_lockless/100 — Same as burst, with the context in single-threaded lockless mode; N is the burst size. ++6% +burst_lockless/100 +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +-10% +concurrent/1 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/1 +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/16 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/16 +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 — N independent accept loops run on separate listeners at once, each doing a one-way one-byte transfer (client writes, server reads) per connection; N is the number of loops. +concurrent/4 +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +sequential — Single connect/accept/close loop with a one-way one-byte transfer (client writes, server reads) per connection, one connection at a time. +-4% +sequential +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +sequential_lockless — Same as sequential, with the context in single-threaded lockless mode. +sequential_lockless + + diff --git a/doc/modules/ROOT/images/bench/windows-fan_out-g1.svg b/doc/modules/ROOT/images/bench/windows-fan_out-g1.svg new file mode 100644 index 000000000..a6d4266ed --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-fan_out-g1.svg @@ -0,0 +1,50 @@ + +fan_out — Windows, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-15% + +-10% + +-5% + ++5% +0% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-7% +concurrent_parents/1 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-11% +concurrent_parents/1 +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-11% +concurrent_parents/16 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-11% +concurrent_parents/16 +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-9% +concurrent_parents/4 — N independent parents each fan out to 16 sub-requests at once; N is the number of parents. +-10% +concurrent_parents/4 +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-7% +concurrent_parents_lockless/1 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-10% +concurrent_parents_lockless/1 +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-11% +concurrent_parents_lockless/16 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-11% +concurrent_parents_lockless/16 +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-9% +concurrent_parents_lockless/4 — Same as concurrent_parents, with the context in single-threaded lockless mode; N is the number of parents. +-11% +concurrent_parents_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/windows-fan_out-g2.svg b/doc/modules/ROOT/images/bench/windows-fan_out-g2.svg new file mode 100644 index 000000000..aa31ed37c --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-fan_out-g2.svg @@ -0,0 +1,62 @@ + +fan_out — Windows, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-20% + +-15% + +-10% + +-5% + ++5% +0% +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/1 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/1 +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/16 +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +-5% +fork_join/4 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/4 +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 — One parent starts N sub-requests to echo servers and waits for all to finish before repeating; N is the fan-out width. +fork_join/64 +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/1 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/1 +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/16 +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/4 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +-10% +fork_join_lockless/4 +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/64 — Same as fork_join, with the context in single-threaded lockless mode; N is the fan-out width. +fork_join_lockless/64 +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/16 +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 — Two-level fan-out: the parent starts N groups of 4 sub-requests each, each group awaited via its own latch; N is the number of groups. +nested/4 +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +-11% +nested_lockless/16 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/16 +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +nested_lockless/4 — Same as nested, with the context in single-threaded lockless mode; N is the number of groups. +-13% +nested_lockless/4 + + diff --git a/doc/modules/ROOT/images/bench/windows-http_server.svg b/doc/modules/ROOT/images/bench/windows-http_server.svg new file mode 100644 index 000000000..ed95e61ce --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-http_server.svg @@ -0,0 +1,57 @@ + +http_server — Windows + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-7.5% + +-5% + +-2.5% + ++2.5% +0% +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/1 +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/16 +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/32 +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 — N client/server pairs run the single_conn request/response loop concurrently; N is the number of connections. +concurrent/4 +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/1 +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +-6% +multithread/16 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +-2% +multithread/16 +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/2 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/2 +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/4 +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 — 32 client/server pairs share one context serviced by N threads running the context; N is the thread count. +multithread/8 +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn — One client repeatedly sends a fixed small HTTP request to one server and reads the response. +single_conn +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +0% +single_conn_lockless — Same as single_conn, with the context in single-threaded lockless mode. +-5% +single_conn_lockless + + diff --git a/doc/modules/ROOT/images/bench/windows-socket_latency.svg b/doc/modules/ROOT/images/bench/windows-socket_latency.svg new file mode 100644 index 000000000..14f90cc03 --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-socket_latency.svg @@ -0,0 +1,60 @@ + +socket_latency — Windows + + +↑ faster (lower mean latency) · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-15% + +-10% + +-5% + ++5% +0% +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/1 +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +-11% +concurrent/16 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +-8% +concurrent/16 +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 — N independent 64-byte pingpong pairs run concurrently on one context; N is the number of connection pairs. +concurrent/4 +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/1 +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/16 +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 — Same as concurrent, with the context in single-threaded lockless mode; N is the number of connection pairs. +concurrent_lockless/4 +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +-6% +pingpong/1 +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/1024 +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 — One TCP connection ping-pongs a message client->server->client; N is the message size in bytes. +pingpong/64 +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1 +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/1024 +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +-7% +pingpong_lockless/64 — Same as pingpong, with the context in single-threaded lockless mode; N is the message size in bytes. +pingpong_lockless/64 + + diff --git a/doc/modules/ROOT/images/bench/windows-socket_throughput-g1.svg b/doc/modules/ROOT/images/bench/windows-socket_throughput-g1.svg new file mode 100644 index 000000000..42c5b9e02 --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-socket_throughput-g1.svg @@ -0,0 +1,60 @@ + +socket_throughput — Windows, part 1 + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-15% + +-10% + +-5% + ++5% +0% +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1024 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +-5% +bidirectional/1024 +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +-9% +bidirectional/1048576 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/1048576 +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/16384 +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. ++0% +bidirectional/262144 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/262144 +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/4096 +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 — Both ends of the connection write and read simultaneously; N is the chunk size in bytes. +bidirectional/65536 +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1024 +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/1048576 +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/16384 +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/262144 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-1% +bidirectional_lockless/262144 +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/4096 +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 — Same as bidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +bidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/windows-socket_throughput-g2.svg b/doc/modules/ROOT/images/bench/windows-socket_throughput-g2.svg new file mode 100644 index 000000000..9c012f2ce --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-socket_throughput-g2.svg @@ -0,0 +1,33 @@ + +socket_throughput — Windows, part 2 + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-4% + +-2% + ++2% +0% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-2% +multithread/2 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-2% +multithread/2 +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-3% +multithread/4 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-3% +multithread/4 +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +0% +multithread/8 — 32 bidirectional connection pairs share one context serviced by N threads running the context, with 64 KiB chunks; N is the thread count. +-1% +multithread/8 + + diff --git a/doc/modules/ROOT/images/bench/windows-socket_throughput-g3.svg b/doc/modules/ROOT/images/bench/windows-socket_throughput-g3.svg new file mode 100644 index 000000000..1d9566a1b --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-socket_throughput-g3.svg @@ -0,0 +1,62 @@ + +socket_throughput — Windows, part 3 + + +↑ faster · dotted line = asio (callbacks) · backend = IOCP + +corosio + +asio (coroutines) + +-6% + +-4% + +-2% + ++2% + ++4% +0% +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1024 +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/1048576 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. ++1% +unidirectional/1048576 +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/16384 +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/262144 +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/4096 +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 — One writer streams to one reader as fast as possible; N is the write/read chunk size in bytes. +unidirectional/65536 +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1024 +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. ++2% +unidirectional_lockless/1048576 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/1048576 +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-3% +unidirectional_lockless/16384 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +-4% +unidirectional_lockless/16384 +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/262144 +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/4096 +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 — Same as unidirectional, with the context in single-threaded lockless mode; N is the chunk size in bytes. +unidirectional_lockless/65536 + + diff --git a/doc/modules/ROOT/images/bench/windows-summary.svg b/doc/modules/ROOT/images/bench/windows-summary.svg new file mode 100644 index 000000000..ac6fb317d --- /dev/null +++ b/doc/modules/ROOT/images/bench/windows-summary.svg @@ -0,0 +1,32 @@ + +Summary — Windows + + +← slower · within noise · faster → per benchmark, vs Boost.Asio (callbacks) + +http_server +8 benchmarks slower than asio (callbacks) +8 +3 benchmarks within noise than asio (callbacks) +3 +-1.7% +socket_throughput +13 benchmarks slower than asio (callbacks) +13 +14 benchmarks within noise than asio (callbacks) +14 +-2.1% +socket_latency +11 benchmarks slower than asio (callbacks) +11 +1 benchmark within noise than asio (callbacks) +-7.3% +accept_churn +9 benchmarks slower than asio (callbacks) +9 +-8.5% +fan_out +18 benchmarks slower than asio (callbacks) +18 +-8.8% + diff --git a/doc/modules/ROOT/nav.adoc b/doc/modules/ROOT/nav.adoc index 2005b0d12..98e1480ac 100644 --- a/doc/modules/ROOT/nav.adoc +++ b/doc/modules/ROOT/nav.adoc @@ -44,6 +44,9 @@ ** xref:5.testing/5a.mocket.adoc[Mock Sockets] ** xref:5.testing/5b.socket-pair.adoc[Socket Pairs] ** xref:5.testing/5c.patterns.adoc[Testing Patterns] -* xref:benchmark-report.adoc[Benchmarks] +* xref:benchmarks/index.adoc[Benchmarks] +** xref:benchmarks/linux.adoc[Linux] +** xref:benchmarks/windows.adoc[Windows] +** xref:benchmarks/macos.adoc[macOS] * xref:glossary.adoc[Glossary] * xref:reference:boost/corosio.adoc[Reference] diff --git a/doc/modules/ROOT/pages/benchmark-report.adoc b/doc/modules/ROOT/pages/benchmark-report.adoc deleted file mode 100644 index 745c1fe60..000000000 --- a/doc/modules/ROOT/pages/benchmark-report.adoc +++ /dev/null @@ -1,1194 +0,0 @@ -= Boost.Corosio Performance Benchmarks -:page-mode: explanation -:toc: left -:toclevels: 3 -:source-highlighter: highlightjs - -== Executive Summary - -This report presents comprehensive performance benchmarks on Windows, using the IOCP (I/O Completion Ports) backend. It compares *Boost.Corosio*, *Boost.Asio with coroutines* (`co_spawn`/`use_awaitable`), and *Boost.Asio with callbacks*. The benchmarks cover handler dispatch, socket throughput, socket latency, HTTP server workloads, timers, and connection -churn. - -=== Bottom Line - -Corosio *outperforms Asio coroutines* in handler dispatch (9-50% faster) and *scales dramatically better* under multi-threaded load. It delivers *equivalent performance* in socket I/O, latency, and HTTP server workloads. Asio callbacks achieve the highest raw single-threaded dispatch throughput, but Corosio closes the gap as thread counts increase. - -=== Where Corosio Excels - -* *Multi-threaded handler scaling:* Best scaling of all three — maintains 89% throughput at 8 threads vs 58% (Asio coroutines) and 53% (Asio callbacks) -* *Concurrent post and run:* 46% faster than Asio coroutines (2.35 Mops/s vs 1.61 Mops/s) -* *Interleaved post/run:* 34% faster than Asio coroutines (2.14 Mops/s vs 1.60 Mops/s) -* *HTTP concurrent connections:* 2-7% higher throughput than Asio coroutines - -=== Where Asio Callbacks Leads - -* *Single-threaded handler post:* 51% faster than Corosio (2.59 Mops/s vs 1.71 Mops/s) -* *Bidirectional socket throughput:* 2.6× higher at large buffers (5.74 GB/s vs 2.18 GB/s at 64KB) - -=== Where Asio Has an Edge - -* *Timer schedule/cancel:* 10× faster (35-38 Mops/s vs 3.44 Mops/s). - Historical snapshot only; see the NOTE in the Timer Benchmarks - section (the standalone timer API has since been removed). -* *Bidirectional socket throughput at large buffers:* Asio coroutines 2.5× faster than Corosio - -=== Where They're Equal - -* *Unidirectional socket throughput:* Within 5% across all buffer sizes -* *Socket latency:* Mean within 2%, p99 within 3% -* *HTTP server throughput:* Within 5% at all thread counts -* *Concurrent timer latency:* Identical across all implementations - -=== Key Insights - -[cols="1,2"] -|=== -| Component | Assessment - -| *Handler Dispatch* -| Corosio 9-50% faster than Asio coroutines; Asio callbacks fastest single-threaded - -| *Multi-threaded Scaling* -| Corosio scales best — only implementation to improve at 2 threads - -| *Socket Throughput* -| Equivalent unidirectional; Asio faster bidirectional at large buffers - -| *Socket Latency* -| Equivalent across all three - -| *HTTP Server* -| Equivalent across all three - -| *Timers* -| Asio faster at schedule/cancel; equivalent fire rate and concurrent behavior -|=== - ---- - -== Detailed Results - -=== Handler Dispatch Summary - -[cols="2,1,1,1,1", options="header"] -|=== -| Scenario | Corosio | Asio Coroutines | Asio Callbacks | Winner - -| Single-threaded post -| 1.71 Mops/s -| 1.57 Mops/s -| *2.59 Mops/s* -| *Callbacks* - -| Multi-threaded (8 threads) -| *1.54 Mops/s* -| 1.03 Mops/s -| 1.51 Mops/s -| *Corosio* - -| Interleaved post/run -| 2.14 Mops/s -| 1.60 Mops/s -| *2.88 Mops/s* -| *Callbacks* - -| Concurrent post/run -| 2.35 Mops/s -| 1.61 Mops/s -| *2.58 Mops/s* -| *Callbacks* -|=== - -=== Socket Throughput Summary - -[cols="2,1,1,1,1", options="header"] -|=== -| Scenario | Corosio | Asio Coroutines | Asio Callbacks | Winner - -| Unidirectional 1KB -| *85.68 MB/s* -| 78.63 MB/s -| 77.33 MB/s -| Corosio (+9%) - -| Unidirectional 64KB -| 2.19 GB/s -| 2.24 GB/s -| *2.31 GB/s* -| Tie - -| Bidirectional 1KB -| 84.34 MB/s -| 73.13 MB/s -| *191.75 MB/s* -| Callbacks - -| Bidirectional 64KB -| 2.18 GB/s -| 5.56 GB/s -| *5.74 GB/s* -| Callbacks -|=== - -=== Socket Latency Summary - -[cols="2,1,1,1,1", options="header"] -|=== -| Scenario | Corosio | Asio Coroutines | Asio Callbacks | Winner - -| Ping-pong mean (64B) -| 10.78 μs -| 10.98 μs -| *10.52 μs* -| Tie - -| Ping-pong p99 (64B) -| 15.00 μs -| 15.10 μs -| *14.70 μs* -| Tie - -| 16 concurrent pairs mean -| 180.64 μs -| 180.71 μs -| *174.83 μs* -| Tie -|=== - -=== HTTP Server Summary - -[cols="2,1,1,1,1", options="header"] -|=== -| Scenario | Corosio | Asio Coroutines | Asio Callbacks | Winner - -| Single connection -| 87.04 Kops/s -| 84.74 Kops/s -| *87.79 Kops/s* -| Tie - -| 32 connections, 8 threads -| 319.24 Kops/s -| 325.73 Kops/s -| *327.99 Kops/s* -| Tie - -| 32 connections, 16 threads -| 422.10 Kops/s -| 422.20 Kops/s -| *426.31 Kops/s* -| Tie -|=== - -=== Timer Summary - -NOTE: These timer figures are a historical snapshot, not current -guidance. The standalone timer API was removed; see the NOTE in the -Timer Benchmarks section. - -[cols="2,1,1,1,1", options="header"] -|=== -| Scenario | Corosio | Asio Coroutines | Asio Callbacks | Winner - -| Schedule/cancel -| 3.44 Mops/s -| 35.73 Mops/s -| *38.05 Mops/s* -| *Asio (10×)* - -| Fire rate -| 110.03 Kops/s -| 118.39 Kops/s -| *119.80 Kops/s* -| Asio (+8%) - -| Concurrent (1000 timers) latency -| 15.45 ms -| *15.39 ms* -| 15.41 ms -| Tie -|=== - -== Test Environment - -[cols="1,3"] -|=== -| Platform | Windows (IOCP backend) -| Duration | 3 seconds per benchmark -| Comparison | Asio coroutines (`co_spawn`/`use_awaitable`) and Asio callbacks -| Measurement | Client-side latency and throughput -|=== - -== Handler Dispatch Benchmarks - -These benchmarks measure raw handler posting and execution throughput, isolating the scheduler from I/O completion overhead. - -=== Single-Threaded Handler Post - -Each implementation posts and runs handlers from a single thread for 3 seconds. - -[cols="1,1,1,1", options="header"] -|=== -| Metric | Corosio | Asio Coroutines | Asio Callbacks - -| Handlers -| 5,134,000 -| 4,712,000 -| 7,764,000 - -| Elapsed -| 3.001 s -| 3.000 s -| 3.000 s - -| *Throughput* -| *1.71 Mops/s* -| 1.57 Mops/s -| *2.59 Mops/s* -|=== - -*Key finding:* Asio callbacks achieve the highest single-threaded dispatch rate. Corosio is 9% faster than Asio coroutines, providing a meaningful advantage for coroutine users. - -=== Multi-Threaded Scaling - -Multiple threads running handlers concurrently. - -[cols="1,1,1,1", options="header"] -|=== -| Threads | Corosio | Asio Coroutines | Asio Callbacks - -| 1 -| 1.72 Mops/s -| 1.78 Mops/s -| *2.82 Mops/s* - -| 2 -| *2.10 Mops/s* (1.23×) -| 1.40 Mops/s (0.78×) -| 2.33 Mops/s (0.83×) - -| 4 -| 2.02 Mops/s (1.18×) -| 1.25 Mops/s (0.70×) -| *2.10 Mops/s* (0.74×) - -| 8 -| *1.54 Mops/s* (0.89×) -| 1.03 Mops/s (0.58×) -| 1.51 Mops/s (0.53×) -|=== - -==== Scaling Analysis - -[role=output] ----- -Throughput vs Thread Count: - -Threads Corosio Asio Coroutine Asio CB Best Scaling - 1 1.72 M 1.78 M 2.82 M — - 2 2.10 M 1.40 M 2.33 M Corosio (1.23×) - 4 2.02 M 1.25 M 2.10 M Corosio (1.18×) - 8 1.54 M 1.03 M 1.51 M Corosio (0.89×) ----- - -*Notable observations:* - -* Corosio is the *only implementation that improves* at 2 threads (1.23× speedup) -* Both Asio approaches degrade immediately at 2 threads (0.78×, 0.83×) -* At 8 threads, Corosio surpasses Asio callbacks despite starting from a lower baseline -* Corosio retains 89% of single-thread throughput at 8 threads, vs 58% (Asio coroutines) and 53% (Asio callbacks) - -=== Interleaved Post/Run - -Alternating between posting batches of 100 handlers and running them. - -[cols="1,1,1,1", options="header"] -|=== -| Metric | Corosio | Asio Coroutines | Asio Callbacks - -| Handlers/iter -| 100 -| 100 -| 100 - -| Total handlers -| 6,408,000 -| 4,792,100 -| 8,651,900 - -| Elapsed -| 3.000 s -| 3.000 s -| 3.000 s - -| *Throughput* -| *2.14 Mops/s* -| 1.60 Mops/s -| *2.88 Mops/s* -|=== - -*Key finding:* Corosio is 34% faster than Asio coroutines in this common real-world pattern. - -=== Concurrent Post and Run - -Four threads simultaneously posting and running handlers. - -[cols="1,1,1,1", options="header"] -|=== -| Metric | Corosio | Asio Coroutines | Asio Callbacks - -| Threads -| 4 -| 4 -| 4 - -| Total handlers -| 7,130,000 -| 4,870,000 -| 7,830,000 - -| Elapsed -| 3.029 s -| 3.024 s -| 3.030 s - -| *Throughput* -| *2.35 Mops/s* -| 1.61 Mops/s -| *2.58 Mops/s* -|=== - -*Key finding:* Corosio is 46% faster than Asio coroutines and within 9% of Asio callbacks in this multi-producer scenario. - -== Socket Throughput Benchmarks - -=== Unidirectional Throughput - -Single direction transfer with varying buffer sizes. - -[cols="1,1,1,1", options="header"] -|=== -| Buffer Size | Corosio | Asio Coroutines | Asio Callbacks - -| 1024 bytes -| *85.68 MB/s* -| 78.63 MB/s -| 77.33 MB/s - -| 4096 bytes -| 259.30 MB/s -| 265.84 MB/s -| *291.03 MB/s* - -| 16384 bytes -| 956.58 MB/s -| 947.64 MB/s -| *997.23 MB/s* - -| 65536 bytes -| 2.19 GB/s -| 2.24 GB/s -| *2.31 GB/s* -|=== - -*Observation:* Unidirectional throughput is within 13% across all three implementations. Corosio has a slight edge at the smallest buffer size. All three are bounded by the same kernel socket path. - -=== Bidirectional Throughput - -Simultaneous transfer in both directions. - -[cols="1,1,1,1", options="header"] -|=== -| Buffer Size | Corosio | Asio Coroutines | Asio Callbacks - -| 1024 bytes -| 84.34 MB/s -| 73.13 MB/s -| *191.75 MB/s* - -| 4096 bytes -| 258.49 MB/s -| 401.06 MB/s -| *674.75 MB/s* - -| 16384 bytes -| 979.91 MB/s -| 2.20 GB/s -| *2.33 GB/s* - -| 65536 bytes -| 2.18 GB/s -| 5.56 GB/s -| *5.74 GB/s* -|=== - -*Observation:* Bidirectional throughput at larger buffer sizes reveals a gap. Corosio's combined bidirectional throughput is comparable to its unidirectional throughput, while both Asio implementations scale beyond their unidirectional numbers. At 64KB, Asio achieves 2.5-2.6× higher bidirectional throughput than Corosio. - -== Socket Latency Benchmarks - -=== Ping-Pong Round-Trip Latency - -A single socket pair exchanges messages for 3 seconds. - -[cols="1,1,1,1", options="header"] -|=== -| Message Size | Corosio Mean | Asio Coroutines Mean | Asio Callbacks Mean - -| 1 byte -| 10.75 μs -| 10.90 μs -| *10.56 μs* - -| 64 bytes -| 10.78 μs -| 10.98 μs -| *10.52 μs* - -| 1024 bytes -| 11.05 μs -| 11.09 μs -| *10.79 μs* -|=== - -==== Latency Distribution (64-byte messages) - -[cols="1,1,1,1", options="header"] -|=== -| Percentile | Corosio | Asio Coroutines | Asio Callbacks - -| p50 -| 10.40 μs -| 10.60 μs -| *10.20 μs* - -| p90 -| 10.70 μs -| 10.80 μs -| *10.40 μs* - -| p99 -| 15.00 μs -| 15.10 μs -| *14.70 μs* - -| p99.9 -| 119.50 μs -| 128.67 μs -| *110.56 μs* - -| min -| *9.10 μs* -| 9.20 μs -| 9.40 μs - -| max -| *1.98 ms* -| 1.22 ms -| *927.80 μs* -|=== - -*Observation:* All three implementations deliver mean, p50, p90, and p99 latency within 5% of each other. Those differences are small enough to be within measurement noise. The far tail spreads wider: p99.9 varies by 16% and max by more than 2×. Asio callbacks has the lowest p99.9 and max. - -=== Concurrent Socket Pairs - -Multiple socket pairs operating concurrently (64-byte messages). - -[cols="1,1,1,1,1,1,1", options="header"] -|=== -| Pairs | Corosio Mean | Asio Coroutine Mean | Asio CB Mean | Corosio p99 | Asio Coroutine p99 | Asio CB p99 - -| 1 -| 10.78 μs -| 10.94 μs -| *10.57 μs* -| 15.30 μs -| 15.30 μs -| *14.70 μs* - -| 4 -| 44.71 μs -| 45.04 μs -| *43.46 μs* -| 94.00 μs -| 93.23 μs -| *87.97 μs* - -| 16 -| 180.64 μs -| 180.71 μs -| *174.83 μs* -| 377.77 μs -| *353.27 μs* -| 368.23 μs -|=== - -*Observation:* All three implementations scale similarly. Asio callbacks has a marginal edge in mean latency. At 16 pairs, Asio coroutines has slightly better p99. - -== HTTP Server Benchmarks - -=== Single Connection (Sequential Requests) - -[cols="1,1,1,1", options="header"] -|=== -| Metric | Corosio | Asio Coroutines | Asio Callbacks - -| Completed -| 261,715 -| 255,257 -| *264,158* - -| *Throughput* -| *87.04 Kops/s* -| 84.74 Kops/s -| *87.79 Kops/s* - -| Mean latency -| 11.46 μs -| 11.76 μs -| *11.36 μs* - -| p99 latency -| 16.30 μs -| 16.30 μs -| *15.90 μs* -|=== - -*Observation:* Single-connection HTTP performance is comparable across all three. Corosio and Asio callbacks are within 1%. - -=== Concurrent Connections (Single Thread) - -[cols="1,1,1,1,1,1,1", options="header"] -|=== -| Connections | Corosio Throughput | Asio Coroutine Throughput | Asio CB Throughput | Corosio Mean | Asio Coroutine Mean | Asio CB Mean - -| 1 -| *86.79 Kops/s* -| 81.50 Kops/s -| 85.65 Kops/s -| 11.49 μs -| 12.24 μs -| *11.65 μs* - -| 4 -| *85.34 Kops/s* -| 80.11 Kops/s -| 83.02 Kops/s -| 46.84 μs -| 49.85 μs -| *48.15 μs* - -| 16 -| *83.40 Kops/s* -| 79.30 Kops/s -| 82.80 Kops/s -| *191.79 μs* -| 201.13 μs -| 193.20 μs - -| 32 -| 80.07 Kops/s -| 78.47 Kops/s -| *81.71 Kops/s* -| 399.56 μs -| 406.99 μs -| *391.54 μs* -|=== - -*Observation:* Corosio consistently outperforms Asio coroutines by 2-7% in concurrent connection throughput. The gap narrows at higher connection counts. Corosio and Asio callbacks trade the lead depending on connection count. - -=== Multi-Threaded HTTP (32 Connections) - -[cols="1,1,1,1", options="header"] -|=== -| Threads | Corosio Throughput | Asio Coroutines Throughput | Asio Callbacks Throughput - -| 1 -| 81.31 Kops/s -| 77.49 Kops/s -| *83.36 Kops/s* - -| 2 -| 115.80 Kops/s -| 114.29 Kops/s -| *118.18 Kops/s* - -| 4 -| 196.40 Kops/s -| 194.05 Kops/s -| *201.64 Kops/s* - -| 8 -| 319.24 Kops/s -| 325.73 Kops/s -| *327.99 Kops/s* - -| 16 -| 422.10 Kops/s -| 422.20 Kops/s -| *426.31 Kops/s* -|=== - -==== Multi-Threaded Latency - -[cols="1,1,1,1,1,1,1", options="header"] -|=== -| Threads | Corosio Mean | Asio Coroutine Mean | Asio CB Mean | Corosio p99 | Asio Coroutine p99 | Asio CB p99 - -| 1 -| 393.50 μs -| 412.09 μs -| *383.85 μs* -| 656.65 μs -| 730.44 μs -| *682.81 μs* - -| 2 -| 276.23 μs -| 279.53 μs -| *270.69 μs* -| 424.65 μs -| 509.19 μs -| *423.52 μs* - -| 4 -| 162.81 μs -| 163.85 μs -| *158.52 μs* -| 230.55 μs -| 230.66 μs -| *224.11 μs* - -| 8 -| 100.10 μs -| *97.77 μs* -| 97.44 μs -| 139.12 μs -| *134.07 μs* -| 144.19 μs - -| 16 -| 75.61 μs -| 75.33 μs -| *74.57 μs* -| 99.86 μs -| *94.40 μs* -| 94.93 μs -|=== - -*Key finding:* All three implementations converge at high thread counts, reaching ~422-426 Kops/s at 16 threads. Both show excellent near-linear scaling. Corosio has slightly higher mean latency at lower thread counts but converges at 8+ threads. - -== Timer Benchmarks - -NOTE: The timer category was removed from the benchmark suite. Corosio's -stateless delay() and timeout() acquire their timer per wait, while Asio -reuses a caller-owned timer object. The two sides therefore no longer -measure comparable work. The figures below are a historical snapshot -taken while both libraries exposed an equivalent timer object API. - -NOTE: The fan-out join primitives are likewise no longer like-for-like: -corosio's `async_waker` latch versus Asio's parked-timer cancel. Those -figures compare each library's idiomatic completion signal rather than -identical mechanisms. - -=== Timer Schedule/Cancel - -Measures the rate of creating and cancelling timers without firing them. - -[cols="1,1,1,1", options="header"] -|=== -| Metric | Corosio | Asio Coroutines | Asio Callbacks - -| Timers -| 10,328,000 -| 107,190,000 -| 114,149,000 - -| Elapsed -| 3.000 s -| 3.000 s -| 3.000 s - -| *Throughput* -| 3.44 Mops/s -| 35.73 Mops/s -| *38.05 Mops/s* -|=== - -*Observation:* Asio is approximately 10× faster at scheduling and cancelling timers. This benchmark isolates the timer data structure operations without involving I/O completion. - -=== Timer Fire Rate - -Measures the rate of timers that actually expire and fire their handlers. - -[cols="1,1,1,1", options="header"] -|=== -| Metric | Corosio | Asio Coroutines | Asio Callbacks - -| Fires -| 331,398 -| 356,602 -| 361,523 - -| Elapsed -| 3.012 s -| 3.012 s -| 3.018 s - -| *Throughput* -| 110.03 Kops/s -| 118.39 Kops/s -| *119.80 Kops/s* -|=== - -*Observation:* When timers actually fire, the gap narrows to ~8%. The bottleneck shifts from the timer data structure to the I/O completion mechanism. - -=== Concurrent Timers - -Multiple timers firing at 15 ms intervals concurrently. - -[cols="1,1,1,1,1,1,1", options="header"] -|=== -| Timers | Corosio Mean | Asio Coroutine Mean | Asio CB Mean | Corosio p99 | Asio Coroutine p99 | Asio CB p99 - -| 10 -| 15.39 ms -| *15.40 ms* -| 15.42 ms -| 18.23 ms -| *16.89 ms* -| 17.29 ms - -| 100 -| 15.43 ms -| *15.40 ms* -| *15.40 ms* -| 17.02 ms -| *16.59 ms* -| 17.61 ms - -| 1000 -| 15.45 ms -| *15.39 ms* -| 15.41 ms -| *16.71 ms* -| 17.47 ms -| 18.17 ms -|=== - -*Observation:* Concurrent timer latency is identical across all three implementations. Mean latency stays within 0.06 ms across implementations and concurrency levels. Every mean sits about 0.4 ms above the nominal 15 ms interval. Corosio has the best p99 at 1000 concurrent timers. - -== Connection Churn Benchmark - -=== Sequential Accept Churn (Corosio) - -Measures the rate of accepting, using, and closing connections sequentially. - -[cols="1,1"] -|=== -| Metric | Value - -| Cycles -| 14,452 - -| Elapsed -| 3.012 s - -| *Throughput* -| *4.80 Kops/s* - -| Mean latency -| 208.28 μs - -| p99 latency -| 457.55 μs - -| Min latency -| 105.40 μs - -| Max latency -| 921.90 μs -|=== - -== Analysis - -=== Handler Dispatch - -The handler dispatch results tell a nuanced story across the three implementations. - -[cols="1,1,1,1", options="header"] -|=== -| Pattern | Corosio vs Asio Coroutine | Corosio vs Asio CB | Notes - -| Single-threaded -| +9% -| -34% -| Callbacks benefit from lower per-handler overhead - -| Multi-threaded (8T) -| +49% -| +2% -| Corosio's scaling advantage closes the gap - -| Interleaved -| +34% -| -26% -| Common real-world pattern - -| Concurrent -| +46% -| -9% -| Multi-producer scenario -|=== - -The most telling result is multi-threaded scaling. Every implementation loses throughput as threads increase due to coordination overhead, but Corosio degrades the least: - -[role=output] ----- -Throughput retained at 8 threads (vs 1 thread): - - Corosio: 89% - Asio Coroutines: 58% - Asio Callbacks: 53% ----- - -This makes Corosio the best choice for applications that distribute work across threads. - -=== Socket I/O - -Unidirectional socket throughput is equivalent across all three implementations, confirming that the kernel socket path — not the user-space framework — is the bottleneck. - -Bidirectional throughput reveals a difference: Asio implementations achieve significantly higher combined throughput at larger buffer sizes. Corosio's bidirectional throughput is comparable to its unidirectional throughput, suggesting serialization between the read and write paths. This is an area for future optimization. - -=== Socket Latency - -Latency results are tightly clustered across all three. Mean latencies differ by less than 0.5 μs. Tail latencies (p99) differ by less than 0.4 μs at the single-pair level. These differences are within measurement noise. - -=== HTTP Server - -HTTP server performance is comparable across all three implementations at all concurrency levels and thread counts. At 16 threads with 32 connections, all three converge to ~422-426 Kops/s. This confirms that for real-world HTTP workloads, the choice of framework has minimal performance impact. - -=== Timers - -Timer schedule/cancel throughput is a notable gap — Asio's timer operations are approximately 10× faster. However, the gap narrows substantially for timer fire rate (8%). It disappears entirely for concurrent timer latency accuracy. Applications that create and cancel timers at very high rates may notice this difference; applications that primarily use timers for timeouts and delays do not. - -=== Summary - -[cols="1,2"] -|=== -| Component | Assessment - -| *Handler Dispatch (vs Asio Coroutine)* -| Corosio 9-50% faster - -| *Handler Dispatch (vs Asio CB)* -| Callbacks faster single-threaded; Corosio matches at 8 threads - -| *Multi-threaded Scaling* -| Corosio best — only one that improves at 2 threads - -| *Socket Throughput (unidirectional)* -| Equivalent - -| *Socket Throughput (bidirectional)* -| Asio 2.5× faster at large buffers - -| *Socket Latency* -| Equivalent - -| *HTTP Throughput* -| Equivalent - -| *Timer Schedule/Cancel* -| Asio 10× faster - -| *Timer Fire/Concurrent* -| Equivalent -|=== - -== Conclusions - -=== Summary - -Corosio delivers *equivalent or better performance* compared to Asio coroutines across the majority of benchmarks: - -* *Handler dispatch:* Corosio is 9-50% faster than Asio coroutines -* *Multi-threaded scaling:* Corosio retains 89% throughput at 8 threads vs 58% for Asio coroutines -* *Socket I/O:* Equivalent unidirectional throughput, equivalent latency -* *HTTP server:* Equivalent throughput and latency -* *Bidirectional throughput:* Asio faster at large buffers — area for optimization -* *Timer schedule/cancel:* Asio faster — area for optimization - -Asio callbacks achieve the highest raw single-threaded dispatch rate, but this advantage diminishes under multi-threaded load where Corosio matches or exceeds it. - -=== Recommendations - -[cols="1,2"] -|=== -| Workload | Recommendation - -| Handler-intensive (single-threaded) -| Asio callbacks fastest; Corosio 9% faster than Asio coroutines - -| Handler-intensive (multi-threaded) -| *Corosio* scales best - -| Socket I/O (unidirectional) -| All equivalent - -| Socket I/O (bidirectional, large buffers) -| *Asio* currently faster - -| HTTP servers -| All equivalent - -| Timer-heavy workloads -| *Asio* faster at schedule/cancel; equivalent for firing -|=== - -=== Key Takeaway - -For coroutine-based async programming on Windows (IOCP), *Corosio provides equivalent or better performance* than Asio coroutines in every category but two. The exceptions are bidirectional socket throughput and timer schedule/cancel, a measurement that predates removal of the standalone timer API. Corosio's superior multi-threaded scaling makes it particularly well-suited for applications that distribute work across threads. Bidirectional throughput and timer operations are identified areas for future -optimization. - -== Appendix: Raw Data - -=== Corosio Results - -[role=output] ----- -Backend: iocp -Duration: 3 s per benchmark - -=== Single-threaded Handler Post (Corosio) === - Handlers: 5134000 - Elapsed: 3.001 s - Throughput: 1.71 Mops/s - -=== Multi-threaded Scaling (Corosio) === - 1 thread(s): 1.72 Mops/s - 2 thread(s): 2.10 Mops/s (speedup: 1.23x) - 4 thread(s): 2.02 Mops/s (speedup: 1.18x) - 8 thread(s): 1.54 Mops/s (speedup: 0.89x) - -=== Interleaved Post/Run (Corosio) === - Handlers/iter: 100 - Total handlers: 6408000 - Elapsed: 3.000 s - Throughput: 2.14 Mops/s - -=== Concurrent Post and Run (Corosio) === - Threads: 4 - Total handlers: 7130000 - Elapsed: 3.029 s - Throughput: 2.35 Mops/s - -=== Unidirectional Throughput (Corosio) === - Buffer size: 1024 bytes: 85.68 MB/s - Buffer size: 4096 bytes: 259.30 MB/s - Buffer size: 16384 bytes: 956.58 MB/s - Buffer size: 65536 bytes: 2.19 GB/s - -=== Bidirectional Throughput (Corosio) === - Buffer size: 1024 bytes: 84.34 MB/s (combined) - Buffer size: 4096 bytes: 258.49 MB/s (combined) - Buffer size: 16384 bytes: 979.91 MB/s (combined) - Buffer size: 65536 bytes: 2.18 GB/s (combined) - -=== Ping-Pong Round-Trip Latency (Corosio) === - 1 byte: mean=10.75 us, p50=10.30 us, p99=15.00 us - 64 bytes: mean=10.78 us, p50=10.40 us, p99=15.00 us - 1024 bytes: mean=11.05 us, p50=10.60 us, p99=15.30 us - -=== Concurrent Socket Pairs Latency (Corosio) === - 1 pair: mean=10.78 us, p99=15.30 us - 4 pairs: mean=44.71 us, p99=94.00 us - 16 pairs: mean=180.64 us, p99=377.77 us - -=== HTTP Single Connection (Corosio) === - Throughput: 87.04 Kops/s - Latency: mean=11.46 us, p99=16.30 us - -=== HTTP Concurrent Connections (Corosio, single thread) === - 1 conn: 86.79 Kops/s, mean=11.49 us, p99=16.60 us - 4 conns: 85.34 Kops/s, mean=46.84 us, p99=105.41 us - 16 conns: 83.40 Kops/s, mean=191.79 us, p99=403.74 us - 32 conns: 80.07 Kops/s, mean=399.56 us, p99=679.69 us - -=== HTTP Multi-threaded (Corosio, 32 connections) === - 1 thread: 81.31 Kops/s, mean=393.50 us, p99=656.65 us - 2 threads: 115.80 Kops/s, mean=276.23 us, p99=424.65 us - 4 threads: 196.40 Kops/s, mean=162.81 us, p99=230.55 us - 8 threads: 319.24 Kops/s, mean=100.10 us, p99=139.12 us - 16 threads: 422.10 Kops/s, mean=75.61 us, p99=99.86 us - -=== Timer Schedule/Cancel (Corosio) === - Timers: 10328000, Throughput: 3.44 Mops/s - -=== Timer Fire Rate (Corosio) === - Fires: 331398, Throughput: 110.03 Kops/s - -=== Concurrent Timers (Corosio) === - 10 timers: mean=15.39 ms, p99=18.23 ms - 100 timers: mean=15.43 ms, p99=17.02 ms - 1000 timers: mean=15.45 ms, p99=16.71 ms - -=== Sequential Accept Churn (Corosio) === - Cycles: 14452, Throughput: 4.80 Kops/s - Latency: mean=208.28 us, p99=457.55 us ----- - -=== Asio Coroutines Results - -[role=output] ----- -=== Single-threaded Handler Post (Asio Coroutines) === - Handlers: 4712000 - Elapsed: 3.000 s - Throughput: 1.57 Mops/s - -=== Multi-threaded Scaling (Asio Coroutines) === - 1 thread(s): 1.78 Mops/s - 2 thread(s): 1.40 Mops/s (speedup: 0.78x) - 4 thread(s): 1.25 Mops/s (speedup: 0.70x) - 8 thread(s): 1.03 Mops/s (speedup: 0.58x) - -=== Interleaved Post/Run (Asio Coroutines) === - Handlers/iter: 100 - Total handlers: 4792100 - Elapsed: 3.000 s - Throughput: 1.60 Mops/s - -=== Concurrent Post and Run (Asio Coroutines) === - Threads: 4 - Total handlers: 4870000 - Elapsed: 3.024 s - Throughput: 1.61 Mops/s - -=== Unidirectional Throughput (Asio Coroutines) === - Buffer size: 1024 bytes: 78.63 MB/s - Buffer size: 4096 bytes: 265.84 MB/s - Buffer size: 16384 bytes: 947.64 MB/s - Buffer size: 65536 bytes: 2.24 GB/s - -=== Bidirectional Throughput (Asio Coroutines) === - Buffer size: 1024 bytes: 73.13 MB/s (combined) - Buffer size: 4096 bytes: 401.06 MB/s (combined) - Buffer size: 16384 bytes: 2.20 GB/s (combined) - Buffer size: 65536 bytes: 5.56 GB/s (combined) - -=== Ping-Pong Round-Trip Latency (Asio Coroutines) === - 1 byte: mean=10.90 us, p50=10.50 us, p99=15.10 us - 64 bytes: mean=10.98 us, p50=10.60 us, p99=15.10 us - 1024 bytes: mean=11.09 us, p50=10.50 us, p99=15.30 us - -=== Concurrent Socket Pairs Latency (Asio Coroutines) === - 1 pair: mean=10.94 us, p99=15.30 us - 4 pairs: mean=45.04 us, p99=93.23 us - 16 pairs: mean=180.71 us, p99=353.27 us - -=== HTTP Single Connection (Asio Coroutines) === - Throughput: 84.74 Kops/s - Latency: mean=11.76 us, p99=16.30 us - -=== HTTP Concurrent Connections (Asio Coroutines, single thread) === - 1 conn: 81.50 Kops/s, mean=12.24 us, p99=24.10 us - 4 conns: 80.11 Kops/s, mean=49.85 us, p99=104.69 us - 16 conns: 79.30 Kops/s, mean=201.13 us, p99=398.32 us - 32 conns: 78.47 Kops/s, mean=406.99 us, p99=645.61 us - -=== HTTP Multi-threaded (Asio Coroutines, 32 connections) === - 1 thread: 77.49 Kops/s, mean=412.09 us, p99=730.44 us - 2 threads: 114.29 Kops/s, mean=279.53 us, p99=509.19 us - 4 threads: 194.05 Kops/s, mean=163.85 us, p99=230.66 us - 8 threads: 325.73 Kops/s, mean=97.77 us, p99=134.07 us - 16 threads: 422.20 Kops/s, mean=75.33 us, p99=94.40 us - -=== Timer Schedule/Cancel (Asio Coroutines) === - Timers: 107190000, Throughput: 35.73 Mops/s - -=== Timer Fire Rate (Asio Coroutines) === - Fires: 356602, Throughput: 118.39 Kops/s - -=== Concurrent Timers (Asio Coroutines) === - 10 timers: mean=15.40 ms, p99=16.89 ms - 100 timers: mean=15.40 ms, p99=16.59 ms - 1000 timers: mean=15.39 ms, p99=17.47 ms ----- - -=== Asio Callbacks Results - -[role=output] ----- -=== Single-threaded Handler Post (Asio Callbacks) === - Handlers: 7764000 - Elapsed: 3.000 s - Throughput: 2.59 Mops/s - -=== Multi-threaded Scaling (Asio Callbacks) === - 1 thread(s): 2.82 Mops/s - 2 thread(s): 2.33 Mops/s (speedup: 0.83x) - 4 thread(s): 2.10 Mops/s (speedup: 0.74x) - 8 thread(s): 1.51 Mops/s (speedup: 0.53x) - -=== Interleaved Post/Run (Asio Callbacks) === - Handlers/iter: 100 - Total handlers: 8651900 - Elapsed: 3.000 s - Throughput: 2.88 Mops/s - -=== Concurrent Post and Run (Asio Callbacks) === - Threads: 4 - Total handlers: 7830000 - Elapsed: 3.030 s - Throughput: 2.58 Mops/s - -=== Unidirectional Throughput (Asio Callbacks) === - Buffer size: 1024 bytes: 77.33 MB/s - Buffer size: 4096 bytes: 291.03 MB/s - Buffer size: 16384 bytes: 997.23 MB/s - Buffer size: 65536 bytes: 2.31 GB/s - -=== Bidirectional Throughput (Asio Callbacks) === - Buffer size: 1024 bytes: 191.75 MB/s (combined) - Buffer size: 4096 bytes: 674.75 MB/s (combined) - Buffer size: 16384 bytes: 2.33 GB/s (combined) - Buffer size: 65536 bytes: 5.74 GB/s (combined) - -=== Ping-Pong Round-Trip Latency (Asio Callbacks) === - 1 byte: mean=10.56 us, p50=10.30 us, p99=14.70 us - 64 bytes: mean=10.52 us, p50=10.20 us, p99=14.70 us - 1024 bytes: mean=10.79 us, p50=10.40 us, p99=15.10 us - -=== Concurrent Socket Pairs Latency (Asio Callbacks) === - 1 pair: mean=10.57 us, p99=14.70 us - 4 pairs: mean=43.46 us, p99=87.97 us - 16 pairs: mean=174.83 us, p99=368.23 us - -=== HTTP Single Connection (Asio Callbacks) === - Throughput: 87.79 Kops/s - Latency: mean=11.36 us, p99=15.90 us - -=== HTTP Concurrent Connections (Asio Callbacks, single thread) === - 1 conn: 85.65 Kops/s, mean=11.65 us, p99=19.40 us - 4 conns: 83.02 Kops/s, mean=48.15 us, p99=106.16 us - 16 conns: 82.80 Kops/s, mean=193.20 us, p99=361.47 us - 32 conns: 81.71 Kops/s, mean=391.54 us, p99=638.11 us - -=== HTTP Multi-threaded (Asio Callbacks, 32 connections) === - 1 thread: 83.36 Kops/s, mean=383.85 us, p99=682.81 us - 2 threads: 118.18 Kops/s, mean=270.69 us, p99=423.52 us - 4 threads: 201.64 Kops/s, mean=158.52 us, p99=224.11 us - 8 threads: 327.99 Kops/s, mean=97.44 us, p99=144.19 us - 16 threads: 426.31 Kops/s, mean=74.57 us, p99=94.93 us - -=== Timer Schedule/Cancel (Asio Callbacks) === - Timers: 114149000, Throughput: 38.05 Mops/s - -=== Timer Fire Rate (Asio Callbacks) === - Fires: 361523, Throughput: 119.80 Kops/s - -=== Concurrent Timers (Asio Callbacks) === - 10 timers: mean=15.42 ms, p99=17.29 ms - 100 timers: mean=15.40 ms, p99=17.61 ms - 1000 timers: mean=15.41 ms, p99=18.17 ms ----- diff --git a/doc/modules/ROOT/pages/benchmarks/index.adoc b/doc/modules/ROOT/pages/benchmarks/index.adoc new file mode 100644 index 000000000..728bbcac5 --- /dev/null +++ b/doc/modules/ROOT/pages/benchmarks/index.adoc @@ -0,0 +1,40 @@ += Boost.Corosio Performance Benchmarks +:page-aliases: benchmark-report.adoc +:page-mode: explanation +:toc: left + +_Generated 2026-10-02 from corosio `e2de06fda069` — https://github.com/cppalliance/corosio/actions/runs/37015995763[benchmark run]. This page is fully generated; do not edit by hand (see <>)._ + +This report compares Boost.Corosio against Boost.Asio using coroutines (the `asio` configuration) and Boost.Asio using callback-based handlers (`asio_callback`) across the corosio benchmark suite. Its purpose is to track corosio's performance relative to an established async I/O library as both evolve, not to crown a universal winner. + +== Methodology + +This run measured 113 benchmarks across 7 categories on Linux; 77 benchmarks across 5 categories on Windows; 113 benchmarks across 7 categories on macOS. + +Each benchmark runs several iterations of a fixed duration, interleaved across configurations. Drift from thermal throttling or background load therefore lands on every configuration alike, rather than skewing one of them. The reported figure for a benchmark/configuration pair is the median across its iterations, which resists a single outlier iteration. The coefficient of variation (CV) across those iterations appears alongside it as a measure of run-to-run noise. See each platform page's Test Environment section for the exact iteration count and duration used there. + +Metric selection and the higher/lower-is-better direction follow the same rules as the PR comparison bot (`.github/bench/compare.py`). Latency categories compare `latency_mean_ns` (lower is better); everything else compares `bytes_per_sec`, then `items_per_sec`, then `ops_per_sec` (higher is better). Percentages on this page are sign-normalized so that a positive value always means corosio performed better than asio. + +Every percentage on these pages is measured against Boost.Asio with callback handlers on the same reactor. Boost.Asio's own coroutine frontend (the `asio` configuration) is plotted alongside corosio against that same baseline. + +On Linux, asio is built twice: once against the epoll reactor, once against io_uring. Every comparison pairs implementations on the same reactor, never judging an io_uring corosio backend against an epoll-only asio. On Windows and macOS, asio uses its native reactor (IOCP and kqueue respectively), matching corosio's own default backend there. Each platform page records the reactor used on that machine. + +Each platform's numbers come from a dedicated, otherwise-idle machine, to keep competing workloads from adding noise to the measurements. + +== Reproducing + +The full suite layout and exact commands are documented in `bench/README.md`. In short: dispatch the benchmark-report workflow, download its artifacts, and regenerate this page with: + +[source,bash,role=external] +---- +python3 .github/bench/report_page.py --input-dir \ + --output-dir doc/modules/ROOT +---- + +Source data: https://github.com/cppalliance/corosio/actions/runs/37015995763[workflow run]. Regenerate this page with the command above. GitHub retains the run's raw JSON artifacts for 90 days; these pages are the durable record. + +== Platforms + +* xref:benchmarks/linux.adoc[Linux] +* xref:benchmarks/windows.adoc[Windows] +* xref:benchmarks/macos.adoc[macOS] diff --git a/doc/modules/ROOT/pages/benchmarks/linux.adoc b/doc/modules/ROOT/pages/benchmarks/linux.adoc new file mode 100644 index 000000000..14fd563ea --- /dev/null +++ b/doc/modules/ROOT/pages/benchmarks/linux.adoc @@ -0,0 +1,1135 @@ += Linux Benchmarks +:page-mode: explanation +:toc: left + +_Generated 2026-10-02 from corosio `e2de06fda069` — https://github.com/cppalliance/corosio/actions/runs/37015995763[benchmark run]. This page is fully generated; do not edit by hand (see the xref:benchmarks/index.adoc[Benchmarks landing page])._ + +== Summary + +_Within noise_ means the relative difference is within twice the combined run-to-run noise (the root-sum-square of each side's CV). Differences that small are indistinguishable from measurement jitter. See xref:benchmarks/index.adoc[Methodology] for how these figures are computed. + +=== io_uring + +**44 faster · 25 within noise · 44 slower** of 113 benchmarks — median **-0.7%** vs Boost.Asio (callbacks) on the same reactor. + +image::bench/linux-summary-uring.svg[summary,role=bch-block,opts=inline] + +=== epoll + +**33 faster · 32 within noise · 48 slower** of 113 benchmarks — median **-2.4%** vs Boost.Asio (callbacks) on the same reactor. + +image::bench/linux-summary-epoll.svg[summary,role=bch-block,opts=inline] + +== Test Environment + +[cols="1,3"] +|=== +| CPU | AMD EPYC 7302P 16-Core Processor +| Cores | 4 +| RAM (GB) | 31 +| OS | Linux 7.0.0-34-generic +| Kernel/build | #34-Ubuntu SMP PREEMPT_DYNAMIC Wed Sep 2 14:29:37 UTC 2026 +| Compiler | c++ (Ubuntu 15.2.0-16ubuntu1) 15.2.0 +| CMake | `cmake version 4.2.3` +| liburing | 2.14-1 +| Boost commit | 39fa1e9a3491bd099b570935b3f3422065f91b03 +| Asio commit | a7dc25b4cb6c49a6946d86ea20664f1027203225 +| Asio reactor | epoll and io_uring (matched per configuration) +| Capy commit | a372a6b054261f29497ac0d19c2a9533f9eaad40 +| Corosio commit | e2de06fda069607f34bc957cbefcbd9ba12fddca +| Corosio branch | pr/benchmark-report +| Date (UTC) | 2026-10-02 +| Iterations | 7 +| Duration per benchmark (s) | 2.0 +|=== + +== Results + +=== `accept_churn` + +Rate of setting up and tearing down short-lived TCP connections: connect, accept, close. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-accept_churn-uring.svg[accept churn comparison (io_uring),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `burst/10` | corosio (io_uring) | 1.70K ops/s | 1.19% | -29.3% +| `burst/10` | asio (coroutines, io_uring) | 2.38K ops/s | 0.97% | -1.0% +| `burst/10` | asio (callbacks, io_uring) | 2.40K ops/s | 1.23% | baseline (io_uring) +| `burst/100` | corosio (io_uring) | 173.5 ops/s | 1.03% | -29.2% +| `burst/100` | asio (coroutines, io_uring) | 245.5 ops/s | 0.95% | +0.2% +| `burst/100` | asio (callbacks, io_uring) | 244.9 ops/s | 0.78% | baseline (io_uring) +| `burst_lockless/10` | corosio (io_uring) | 1.81K ops/s | 1.38% | -25.0% +| `burst_lockless/10` | asio (coroutines, io_uring) | 2.39K ops/s | 1.12% | -1.0% +| `burst_lockless/10` | asio (callbacks, io_uring) | 2.42K ops/s | 1.20% | baseline (io_uring) +| `burst_lockless/100` | corosio (io_uring) | 185.0 ops/s | 1.10% | -25.1% +| `burst_lockless/100` | asio (coroutines, io_uring) | 248.0 ops/s | 0.88% | +0.4% +| `burst_lockless/100` | asio (callbacks, io_uring) | 246.9 ops/s | 0.70% | baseline (io_uring) +| `concurrent/1` | corosio (io_uring) | 12.66K ops/s | 1.30% | -26.1% +| `concurrent/1` | asio (coroutines, io_uring) | 16.65K ops/s | 0.90% | -2.8% +| `concurrent/1` | asio (callbacks, io_uring) | 17.13K ops/s | 0.93% | baseline (io_uring) +| `concurrent/16` | corosio (io_uring) | 12.44K ops/s | 1.00% | -35.3% +| `concurrent/16` | asio (coroutines, io_uring) | 18.96K ops/s | 0.89% | -1.4% +| `concurrent/16` | asio (callbacks, io_uring) | 19.22K ops/s | 0.96% | baseline (io_uring) +| `concurrent/4` | corosio (io_uring) | 12.54K ops/s | 1.01% | -31.8% +| `concurrent/4` | asio (coroutines, io_uring) | 18.04K ops/s | 0.95% | -1.9% +| `concurrent/4` | asio (callbacks, io_uring) | 18.39K ops/s | 0.93% | baseline (io_uring) +| `sequential` | corosio (io_uring) | 12.68K ops/s | 1.32% | -25.7% +| `sequential` | asio (coroutines, io_uring) | 16.73K ops/s | 0.55% | -2.0% +| `sequential` | asio (callbacks, io_uring) | 17.07K ops/s | 0.88% | baseline (io_uring) +| `sequential_lockless` | corosio (io_uring) | 13.52K ops/s | 1.24% | -21.7% +| `sequential_lockless` | asio (coroutines, io_uring) | 16.86K ops/s | 0.91% | -2.4% +| `sequential_lockless` | asio (callbacks, io_uring) | 17.27K ops/s | 0.87% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-accept_churn-epoll.svg[accept churn comparison (epoll),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `burst/10` | corosio (epoll) | 1.87K ops/s | 0.71% | -12.5% +| `burst/10` | asio (coroutines, epoll) | 2.05K ops/s | 0.26% | -4.2% +| `burst/10` | asio (callbacks, epoll) | 2.14K ops/s | 0.84% | baseline (epoll) +| `burst/100` | corosio (epoll) | 188.4 ops/s | 0.85% | -12.8% +| `burst/100` | asio (coroutines, epoll) | 208.7 ops/s | 0.26% | -3.3% +| `burst/100` | asio (callbacks, epoll) | 215.9 ops/s | 0.24% | baseline (epoll) +| `burst_lockless/10` | corosio (epoll) | 1.88K ops/s | 0.86% | -12.3% +| `burst_lockless/10` | asio (coroutines, epoll) | 2.06K ops/s | 0.91% | -4.1% +| `burst_lockless/10` | asio (callbacks, epoll) | 2.14K ops/s | 0.24% | baseline (epoll) +| `burst_lockless/100` | corosio (epoll) | 189.9 ops/s | 0.78% | -12.3% +| `burst_lockless/100` | asio (coroutines, epoll) | 209.8 ops/s | 0.89% | -3.1% +| `burst_lockless/100` | asio (callbacks, epoll) | 216.5 ops/s | 0.49% | baseline (epoll) +| `concurrent/1` | corosio (epoll) | 14.90K ops/s | 0.38% | -4.7% +| `concurrent/1` | asio (coroutines, epoll) | 15.19K ops/s | 0.35% | -2.9% +| `concurrent/1` | asio (callbacks, epoll) | 15.63K ops/s | 1.13% | baseline (epoll) +| `concurrent/16` | corosio (epoll) | 14.91K ops/s | 0.45% | -9.3% +| `concurrent/16` | asio (coroutines, epoll) | 16.12K ops/s | 0.28% | -1.9% +| `concurrent/16` | asio (callbacks, epoll) | 16.43K ops/s | 0.97% | baseline (epoll) +| `concurrent/4` | corosio (epoll) | 14.91K ops/s | 0.40% | -7.0% +| `concurrent/4` | asio (coroutines, epoll) | 15.66K ops/s | 0.40% | -2.3% +| `concurrent/4` | asio (callbacks, epoll) | 16.03K ops/s | 0.94% | baseline (epoll) +| `sequential` | corosio (epoll) | 14.93K ops/s | 0.48% | -3.7% +| `sequential` | asio (coroutines, epoll) | 15.20K ops/s | 0.98% | -2.0% +| `sequential` | asio (callbacks, epoll) | 15.51K ops/s | 0.58% | baseline (epoll) +| `sequential_lockless` | corosio (epoll) | 14.99K ops/s | 0.51% | -4.7% +| `sequential_lockless` | asio (coroutines, epoll) | 15.34K ops/s | 0.86% | -2.4% +| `sequential_lockless` | asio (callbacks, epoll) | 15.73K ops/s | 0.83% | baseline (epoll) +|=== +==== + +-- + +++++ +
+++++ + +=== `fan_out` + +Fan-out/fan-in coroutine coordination: a parent starts concurrent sub-requests against echo servers and awaits their completion via a shared latch. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-fan_out-uring-g1.svg[fan out comparison (io_uring), part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent_parents/1` | corosio (io_uring) | 3.43K ops/s | 0.77% | -12.6% +| `concurrent_parents/1` | asio (coroutines, io_uring) | 3.78K ops/s | 1.13% | -3.7% +| `concurrent_parents/1` | asio (callbacks, io_uring) | 3.92K ops/s | 0.60% | baseline (io_uring) +| `concurrent_parents/16` | corosio (io_uring) | 3.32K ops/s | 0.47% | -13.3% +| `concurrent_parents/16` | asio (coroutines, io_uring) | 3.71K ops/s | 1.26% | -3.0% +| `concurrent_parents/16` | asio (callbacks, io_uring) | 3.82K ops/s | 0.58% | baseline (io_uring) +| `concurrent_parents/4` | corosio (io_uring) | 3.34K ops/s | 0.49% | -13.3% +| `concurrent_parents/4` | asio (coroutines, io_uring) | 3.73K ops/s | 1.16% | -3.1% +| `concurrent_parents/4` | asio (callbacks, io_uring) | 3.85K ops/s | 0.26% | baseline (io_uring) +| `concurrent_parents_lockless/1` | corosio (io_uring) | 3.65K ops/s | 0.26% | -8.1% +| `concurrent_parents_lockless/1` | asio (coroutines, io_uring) | 3.83K ops/s | 1.27% | -3.5% +| `concurrent_parents_lockless/1` | asio (callbacks, io_uring) | 3.97K ops/s | 0.92% | baseline (io_uring) +| `concurrent_parents_lockless/16` | corosio (io_uring) | 3.53K ops/s | 0.16% | -9.0% +| `concurrent_parents_lockless/16` | asio (coroutines, io_uring) | 3.75K ops/s | 0.85% | -3.3% +| `concurrent_parents_lockless/16` | asio (callbacks, io_uring) | 3.88K ops/s | 0.88% | baseline (io_uring) +| `concurrent_parents_lockless/4` | corosio (io_uring) | 3.55K ops/s | 0.44% | -8.8% +| `concurrent_parents_lockless/4` | asio (coroutines, io_uring) | 3.77K ops/s | 0.84% | -3.3% +| `concurrent_parents_lockless/4` | asio (callbacks, io_uring) | 3.89K ops/s | 0.84% | baseline (io_uring) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-fan_out-uring-g2.svg[fan out comparison (io_uring), part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `fork_join/1` | corosio (io_uring) | 53.35K ops/s | 1.17% | -0.9% +| `fork_join/1` | asio (coroutines, io_uring) | 47.61K ops/s | 1.20% | -11.6% +| `fork_join/1` | asio (callbacks, io_uring) | 53.83K ops/s | 1.02% | baseline (io_uring) +| `fork_join/16` | corosio (io_uring) | 3.44K ops/s | 0.97% | -12.3% +| `fork_join/16` | asio (coroutines, io_uring) | 3.79K ops/s | 0.84% | -3.4% +| `fork_join/16` | asio (callbacks, io_uring) | 3.92K ops/s | 1.20% | baseline (io_uring) +| `fork_join/4` | corosio (io_uring) | 13.74K ops/s | 1.19% | -9.7% +| `fork_join/4` | asio (coroutines, io_uring) | 14.39K ops/s | 0.91% | -5.3% +| `fork_join/4` | asio (callbacks, io_uring) | 15.21K ops/s | 1.19% | baseline (io_uring) +| `fork_join/64` | corosio (io_uring) | 813.2 ops/s | 1.01% | -15.2% +| `fork_join/64` | asio (coroutines, io_uring) | 933.8 ops/s | 0.85% | -2.6% +| `fork_join/64` | asio (callbacks, io_uring) | 958.5 ops/s | 0.88% | baseline (io_uring) +| `fork_join_lockless/1` | corosio (io_uring) | 55.85K ops/s | 0.69% | +1.7% +| `fork_join_lockless/1` | asio (coroutines, io_uring) | 48.68K ops/s | 1.11% | -11.3% +| `fork_join_lockless/1` | asio (callbacks, io_uring) | 54.89K ops/s | 1.12% | baseline (io_uring) +| `fork_join_lockless/16` | corosio (io_uring) | 3.66K ops/s | 0.83% | -7.9% +| `fork_join_lockless/16` | asio (coroutines, io_uring) | 3.85K ops/s | 0.98% | -3.1% +| `fork_join_lockless/16` | asio (callbacks, io_uring) | 3.97K ops/s | 0.92% | baseline (io_uring) +| `fork_join_lockless/4` | corosio (io_uring) | 14.55K ops/s | 0.19% | -5.7% +| `fork_join_lockless/4` | asio (coroutines, io_uring) | 14.57K ops/s | 1.27% | -5.6% +| `fork_join_lockless/4` | asio (callbacks, io_uring) | 15.43K ops/s | 1.08% | baseline (io_uring) +| `fork_join_lockless/64` | corosio (io_uring) | 867.2 ops/s | 0.73% | -10.5% +| `fork_join_lockless/64` | asio (coroutines, io_uring) | 946.7 ops/s | 1.21% | -2.3% +| `fork_join_lockless/64` | asio (callbacks, io_uring) | 969.4 ops/s | 0.83% | baseline (io_uring) +| `nested/16` | corosio (io_uring) | 804.1 ops/s | 1.20% | -16.1% +| `nested/16` | asio (coroutines, io_uring) | 924.9 ops/s | 0.94% | -3.5% +| `nested/16` | asio (callbacks, io_uring) | 958.0 ops/s | 0.59% | baseline (io_uring) +| `nested/4` | corosio (io_uring) | 3.40K ops/s | 1.16% | -13.3% +| `nested/4` | asio (coroutines, io_uring) | 3.77K ops/s | 1.08% | -3.7% +| `nested/4` | asio (callbacks, io_uring) | 3.92K ops/s | 0.90% | baseline (io_uring) +| `nested_lockless/16` | corosio (io_uring) | 856.8 ops/s | 0.83% | -11.7% +| `nested_lockless/16` | asio (coroutines, io_uring) | 935.0 ops/s | 1.11% | -3.7% +| `nested_lockless/16` | asio (callbacks, io_uring) | 970.5 ops/s | 0.62% | baseline (io_uring) +| `nested_lockless/4` | corosio (io_uring) | 3.63K ops/s | 1.08% | -8.7% +| `nested_lockless/4` | asio (coroutines, io_uring) | 3.81K ops/s | 1.04% | -4.1% +| `nested_lockless/4` | asio (callbacks, io_uring) | 3.97K ops/s | 0.65% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-fan_out-epoll-g1.svg[fan out comparison (epoll), part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent_parents/1` | corosio (epoll) | 3.69K ops/s | 0.10% | -8.3% +| `concurrent_parents/1` | asio (coroutines, epoll) | 3.83K ops/s | 1.00% | -4.8% +| `concurrent_parents/1` | asio (callbacks, epoll) | 4.02K ops/s | 0.45% | baseline (epoll) +| `concurrent_parents/16` | corosio (epoll) | 3.59K ops/s | 0.10% | -8.8% +| `concurrent_parents/16` | asio (coroutines, epoll) | 3.74K ops/s | 0.81% | -4.9% +| `concurrent_parents/16` | asio (callbacks, epoll) | 3.93K ops/s | 0.77% | baseline (epoll) +| `concurrent_parents/4` | corosio (epoll) | 3.60K ops/s | 0.12% | -8.6% +| `concurrent_parents/4` | asio (coroutines, epoll) | 3.74K ops/s | 0.97% | -4.8% +| `concurrent_parents/4` | asio (callbacks, epoll) | 3.93K ops/s | 0.83% | baseline (epoll) +| `concurrent_parents_lockless/1` | corosio (epoll) | 3.73K ops/s | 0.12% | -8.2% +| `concurrent_parents_lockless/1` | asio (coroutines, epoll) | 3.89K ops/s | 0.28% | -4.5% +| `concurrent_parents_lockless/1` | asio (callbacks, epoll) | 4.07K ops/s | 0.85% | baseline (epoll) +| `concurrent_parents_lockless/16` | corosio (epoll) | 3.62K ops/s | 0.81% | -9.2% +| `concurrent_parents_lockless/16` | asio (coroutines, epoll) | 3.80K ops/s | 0.19% | -4.8% +| `concurrent_parents_lockless/16` | asio (callbacks, epoll) | 3.99K ops/s | 0.80% | baseline (epoll) +| `concurrent_parents_lockless/4` | corosio (epoll) | 3.63K ops/s | 0.59% | -8.8% +| `concurrent_parents_lockless/4` | asio (coroutines, epoll) | 3.80K ops/s | 0.32% | -4.6% +| `concurrent_parents_lockless/4` | asio (callbacks, epoll) | 3.98K ops/s | 0.87% | baseline (epoll) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-fan_out-epoll-g2.svg[fan out comparison (epoll), part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `fork_join/1` | corosio (epoll) | 54.81K ops/s | 0.95% | -8.6% +| `fork_join/1` | asio (coroutines, epoll) | 52.61K ops/s | 0.98% | -12.3% +| `fork_join/1` | asio (callbacks, epoll) | 59.96K ops/s | 1.01% | baseline (epoll) +| `fork_join/16` | corosio (epoll) | 3.69K ops/s | 0.21% | -8.2% +| `fork_join/16` | asio (coroutines, epoll) | 3.84K ops/s | 0.96% | -4.5% +| `fork_join/16` | asio (callbacks, epoll) | 4.02K ops/s | 0.99% | baseline (epoll) +| `fork_join/4` | corosio (epoll) | 14.65K ops/s | 0.13% | -7.6% +| `fork_join/4` | asio (coroutines, epoll) | 14.84K ops/s | 0.98% | -6.4% +| `fork_join/4` | asio (callbacks, epoll) | 15.85K ops/s | 0.60% | baseline (epoll) +| `fork_join/64` | corosio (epoll) | 900.2 ops/s | 0.12% | -8.0% +| `fork_join/64` | asio (coroutines, epoll) | 939.3 ops/s | 0.87% | -4.0% +| `fork_join/64` | asio (callbacks, epoll) | 978.8 ops/s | 0.92% | baseline (epoll) +| `fork_join_lockless/1` | corosio (epoll) | 55.61K ops/s | 0.21% | -8.6% +| `fork_join_lockless/1` | asio (coroutines, epoll) | 53.42K ops/s | 0.69% | -12.2% +| `fork_join_lockless/1` | asio (callbacks, epoll) | 60.85K ops/s | 0.89% | baseline (epoll) +| `fork_join_lockless/16` | corosio (epoll) | 3.73K ops/s | 0.82% | -8.5% +| `fork_join_lockless/16` | asio (coroutines, epoll) | 3.88K ops/s | 0.26% | -4.7% +| `fork_join_lockless/16` | asio (callbacks, epoll) | 4.08K ops/s | 0.90% | baseline (epoll) +| `fork_join_lockless/4` | corosio (epoll) | 14.84K ops/s | 0.19% | -7.7% +| `fork_join_lockless/4` | asio (coroutines, epoll) | 15.05K ops/s | 0.32% | -6.5% +| `fork_join_lockless/4` | asio (callbacks, epoll) | 16.09K ops/s | 0.86% | baseline (epoll) +| `fork_join_lockless/64` | corosio (epoll) | 910.3 ops/s | 0.88% | -8.2% +| `fork_join_lockless/64` | asio (coroutines, epoll) | 952.6 ops/s | 0.18% | -3.9% +| `fork_join_lockless/64` | asio (callbacks, epoll) | 991.7 ops/s | 0.77% | baseline (epoll) +| `nested/16` | corosio (epoll) | 893.1 ops/s | 0.88% | -8.4% +| `nested/16` | asio (coroutines, epoll) | 929.2 ops/s | 0.23% | -4.7% +| `nested/16` | asio (callbacks, epoll) | 975.4 ops/s | 0.97% | baseline (epoll) +| `nested/4` | corosio (epoll) | 3.66K ops/s | 0.91% | -8.6% +| `nested/4` | asio (coroutines, epoll) | 3.78K ops/s | 0.27% | -5.7% +| `nested/4` | asio (callbacks, epoll) | 4.01K ops/s | 1.03% | baseline (epoll) +| `nested_lockless/16` | corosio (epoll) | 902.4 ops/s | 0.10% | -9.0% +| `nested_lockless/16` | asio (coroutines, epoll) | 943.3 ops/s | 0.86% | -4.9% +| `nested_lockless/16` | asio (callbacks, epoll) | 991.4 ops/s | 0.87% | baseline (epoll) +| `nested_lockless/4` | corosio (epoll) | 3.71K ops/s | 0.89% | -8.9% +| `nested_lockless/4` | asio (coroutines, epoll) | 3.84K ops/s | 0.86% | -5.9% +| `nested_lockless/4` | asio (callbacks, epoll) | 4.08K ops/s | 0.85% | baseline (epoll) +|=== +==== + +-- + +++++ +
+++++ + +=== `http_server` + +Request/response throughput of a minimal HTTP/1.1 server exchanging a fixed small request and canned response over persistent TCP loopback connections. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-http_server-uring.svg[http server comparison (io_uring),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (io_uring) | 53.45K ops/s | 1.33% | +1.3% +| `concurrent/1` | asio (coroutines, io_uring) | 51.57K ops/s | 1.06% | -2.3% +| `concurrent/1` | asio (callbacks, io_uring) | 52.77K ops/s | 1.16% | baseline (io_uring) +| `concurrent/16` | corosio (io_uring) | 52.87K ops/s | 0.99% | -14.0% +| `concurrent/16` | asio (coroutines, io_uring) | 60.33K ops/s | 0.78% | -1.9% +| `concurrent/16` | asio (callbacks, io_uring) | 61.50K ops/s | 0.94% | baseline (io_uring) +| `concurrent/32` | corosio (io_uring) | 52.55K ops/s | 0.83% | -14.5% +| `concurrent/32` | asio (coroutines, io_uring) | 60.23K ops/s | 1.05% | -2.0% +| `concurrent/32` | asio (callbacks, io_uring) | 61.44K ops/s | 0.41% | baseline (io_uring) +| `concurrent/4` | corosio (io_uring) | 53.06K ops/s | 1.22% | -10.9% +| `concurrent/4` | asio (coroutines, io_uring) | 58.30K ops/s | 0.88% | -2.1% +| `concurrent/4` | asio (callbacks, io_uring) | 59.57K ops/s | 0.92% | baseline (io_uring) +| `multithread/1` | corosio (io_uring) | 52.53K ops/s | 0.74% | -14.4% +| `multithread/1` | asio (coroutines, io_uring) | 60.16K ops/s | 0.78% | -2.0% +| `multithread/1` | asio (callbacks, io_uring) | 61.39K ops/s | 0.28% | baseline (io_uring) +| `multithread/16` | corosio (io_uring) | 81.86K ops/s | 13.26% | +90.3% +| `multithread/16` | asio (coroutines, io_uring) | 40.11K ops/s | 1.71% | -6.7% +| `multithread/16` | asio (callbacks, io_uring) | 43.01K ops/s | 1.66% | baseline (io_uring) +| `multithread/2` | corosio (io_uring) | 90.67K ops/s | 0.46% | +91.3% +| `multithread/2` | asio (coroutines, io_uring) | 47.35K ops/s | 8.15% | -0.1% +| `multithread/2` | asio (callbacks, io_uring) | 47.41K ops/s | 10.18% | baseline (io_uring) +| `multithread/4` | corosio (io_uring) | 80.23K ops/s | 0.39% | +72.7% +| `multithread/4` | asio (coroutines, io_uring) | 45.90K ops/s | 0.35% | -1.2% +| `multithread/4` | asio (callbacks, io_uring) | 46.46K ops/s | 0.27% | baseline (io_uring) +| `multithread/8` | corosio (io_uring) | 77.71K ops/s | 0.53% | +78.9% +| `multithread/8` | asio (coroutines, io_uring) | 41.91K ops/s | 0.95% | -3.5% +| `multithread/8` | asio (callbacks, io_uring) | 43.43K ops/s | 0.80% | baseline (io_uring) +| `single_conn` | corosio (io_uring) | 53.68K ops/s | 1.21% | +1.8% +| `single_conn` | asio (coroutines, io_uring) | 51.56K ops/s | 0.98% | -2.3% +| `single_conn` | asio (callbacks, io_uring) | 52.75K ops/s | 0.80% | baseline (io_uring) +| `single_conn_lockless` | corosio (io_uring) | 56.14K ops/s | 1.05% | +4.5% +| `single_conn_lockless` | asio (coroutines, io_uring) | 52.55K ops/s | 1.11% | -2.2% +| `single_conn_lockless` | asio (callbacks, io_uring) | 53.74K ops/s | 1.15% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-http_server-epoll.svg[http server comparison (epoll),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (epoll) | 55.02K ops/s | 0.21% | -6.0% +| `concurrent/1` | asio (coroutines, epoll) | 57.04K ops/s | 1.14% | -2.6% +| `concurrent/1` | asio (callbacks, epoll) | 58.53K ops/s | 0.13% | baseline (epoll) +| `concurrent/16` | corosio (epoll) | 57.82K ops/s | 0.14% | -6.8% +| `concurrent/16` | asio (coroutines, epoll) | 60.14K ops/s | 1.28% | -3.0% +| `concurrent/16` | asio (callbacks, epoll) | 62.02K ops/s | 0.73% | baseline (epoll) +| `concurrent/32` | corosio (epoll) | 57.45K ops/s | 0.13% | -6.6% +| `concurrent/32` | asio (coroutines, epoll) | 59.51K ops/s | 1.16% | -3.2% +| `concurrent/32` | asio (callbacks, epoll) | 61.48K ops/s | 0.83% | baseline (epoll) +| `concurrent/4` | corosio (epoll) | 57.08K ops/s | 0.14% | -7.0% +| `concurrent/4` | asio (coroutines, epoll) | 60.48K ops/s | 1.27% | -1.5% +| `concurrent/4` | asio (callbacks, epoll) | 61.37K ops/s | 0.22% | baseline (epoll) +| `multithread/1` | corosio (epoll) | 57.35K ops/s | 0.59% | -6.7% +| `multithread/1` | asio (coroutines, epoll) | 59.50K ops/s | 1.08% | -3.2% +| `multithread/1` | asio (callbacks, epoll) | 61.45K ops/s | 0.89% | baseline (epoll) +| `multithread/16` | corosio (epoll) | 161.9K ops/s | 0.20% | -1.0% +| `multithread/16` | asio (coroutines, epoll) | 155.4K ops/s | 0.31% | -4.9% +| `multithread/16` | asio (callbacks, epoll) | 163.4K ops/s | 0.53% | baseline (epoll) +| `multithread/2` | corosio (epoll) | 86.93K ops/s | 0.44% | -7.5% +| `multithread/2` | asio (coroutines, epoll) | 89.12K ops/s | 0.51% | -5.2% +| `multithread/2` | asio (callbacks, epoll) | 94.00K ops/s | 0.67% | baseline (epoll) +| `multithread/4` | corosio (epoll) | 160.1K ops/s | 0.47% | -3.4% +| `multithread/4` | asio (coroutines, epoll) | 157.3K ops/s | 0.61% | -5.1% +| `multithread/4` | asio (callbacks, epoll) | 165.8K ops/s | 0.65% | baseline (epoll) +| `multithread/8` | corosio (epoll) | 161.1K ops/s | 0.21% | -1.4% +| `multithread/8` | asio (coroutines, epoll) | 155.7K ops/s | 0.29% | -4.7% +| `multithread/8` | asio (callbacks, epoll) | 163.3K ops/s | 0.52% | baseline (epoll) +| `single_conn` | corosio (epoll) | 55.09K ops/s | 1.04% | -5.9% +| `single_conn` | asio (coroutines, epoll) | 56.90K ops/s | 0.92% | -2.8% +| `single_conn` | asio (callbacks, epoll) | 58.56K ops/s | 0.85% | baseline (epoll) +| `single_conn_lockless` | corosio (epoll) | 55.80K ops/s | 0.46% | -5.9% +| `single_conn_lockless` | asio (coroutines, epoll) | 57.70K ops/s | 1.10% | -2.7% +| `single_conn_lockless` | asio (callbacks, epoll) | 59.30K ops/s | 0.98% | baseline (epoll) +|=== +==== + +-- + +++++ +
+++++ + +=== `local_socket_latency` + +Round-trip latency of a single Unix domain stream socket write/read exchange across message sizes and concurrent pair counts. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-local_socket_latency-uring.svg[local socket latency comparison (io_uring),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (io_uring) | 6.56 µs | 4.09% | +25.3% +| `concurrent/1` | asio (coroutines, io_uring) | 9.22 µs | 2.67% | -5.0% +| `concurrent/1` | asio (callbacks, io_uring) | 8.78 µs | 0.16% | baseline (io_uring) +| `concurrent/16` | corosio (io_uring) | 108.34 µs | 1.05% | -6.5% +| `concurrent/16` | asio (coroutines, io_uring) | 105.53 µs | 0.75% | -3.7% +| `concurrent/16` | asio (callbacks, io_uring) | 101.74 µs | 0.75% | baseline (io_uring) +| `concurrent/4` | corosio (io_uring) | 26.94 µs | 1.58% | +1.8% +| `concurrent/4` | asio (coroutines, io_uring) | 28.20 µs | 1.47% | -2.8% +| `concurrent/4` | asio (callbacks, io_uring) | 27.43 µs | 1.11% | baseline (io_uring) +| `concurrent_lockless/1` | corosio (io_uring) | 6.55 µs | 4.18% | +23.7% +| `concurrent_lockless/1` | asio (coroutines, io_uring) | 8.77 µs | 3.11% | -2.2% +| `concurrent_lockless/1` | asio (callbacks, io_uring) | 8.58 µs | 3.48% | baseline (io_uring) +| `concurrent_lockless/16` | corosio (io_uring) | 108.13 µs | 0.82% | -10.7% +| `concurrent_lockless/16` | asio (coroutines, io_uring) | 102.49 µs | 1.04% | -5.0% +| `concurrent_lockless/16` | asio (callbacks, io_uring) | 97.65 µs | 0.84% | baseline (io_uring) +| `concurrent_lockless/4` | corosio (io_uring) | 27.08 µs | 0.88% | -2.6% +| `concurrent_lockless/4` | asio (coroutines, io_uring) | 27.52 µs | 1.41% | -4.3% +| `concurrent_lockless/4` | asio (callbacks, io_uring) | 26.39 µs | 0.74% | baseline (io_uring) +| `pingpong/1` | corosio (io_uring) | 6.53 µs | 3.23% | +25.5% +| `pingpong/1` | asio (coroutines, io_uring) | 9.08 µs | 2.60% | -3.6% +| `pingpong/1` | asio (callbacks, io_uring) | 8.76 µs | 3.62% | baseline (io_uring) +| `pingpong/1024` | corosio (io_uring) | 7.43 µs | 5.34% | +19.5% +| `pingpong/1024` | asio (coroutines, io_uring) | 9.99 µs | 3.17% | -8.2% +| `pingpong/1024` | asio (callbacks, io_uring) | 9.23 µs | 3.64% | baseline (io_uring) +| `pingpong/64` | corosio (io_uring) | 6.59 µs | 4.50% | +24.9% +| `pingpong/64` | asio (coroutines, io_uring) | 9.18 µs | 1.08% | -4.6% +| `pingpong/64` | asio (callbacks, io_uring) | 8.78 µs | 3.19% | baseline (io_uring) +| `pingpong_lockless/1` | corosio (io_uring) | 6.53 µs | 4.41% | +21.0% +| `pingpong_lockless/1` | asio (coroutines, io_uring) | 8.73 µs | 3.80% | -5.6% +| `pingpong_lockless/1` | asio (callbacks, io_uring) | 8.27 µs | 2.59% | baseline (io_uring) +| `pingpong_lockless/1024` | corosio (io_uring) | 7.45 µs | 4.84% | +19.0% +| `pingpong_lockless/1024` | asio (coroutines, io_uring) | 9.04 µs | 4.01% | +1.7% +| `pingpong_lockless/1024` | asio (callbacks, io_uring) | 9.20 µs | 2.66% | baseline (io_uring) +| `pingpong_lockless/64` | corosio (io_uring) | 6.56 µs | 3.76% | +21.0% +| `pingpong_lockless/64` | asio (coroutines, io_uring) | 8.77 µs | 2.99% | -5.5% +| `pingpong_lockless/64` | asio (callbacks, io_uring) | 8.31 µs | 0.28% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-local_socket_latency-epoll.svg[local socket latency comparison (epoll),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (epoll) | 7.11 µs | 3.77% | +13.4% +| `concurrent/1` | asio (coroutines, epoll) | 8.60 µs | 3.20% | -4.6% +| `concurrent/1` | asio (callbacks, epoll) | 8.22 µs | 3.11% | baseline (epoll) +| `concurrent/16` | corosio (epoll) | 110.13 µs | 0.73% | -4.5% +| `concurrent/16` | asio (coroutines, epoll) | 112.84 µs | 0.71% | -7.1% +| `concurrent/16` | asio (callbacks, epoll) | 105.39 µs | 0.75% | baseline (epoll) +| `concurrent/4` | corosio (epoll) | 27.52 µs | 1.65% | +1.0% +| `concurrent/4` | asio (coroutines, epoll) | 29.61 µs | 1.67% | -6.5% +| `concurrent/4` | asio (callbacks, epoll) | 27.79 µs | 1.31% | baseline (epoll) +| `concurrent_lockless/1` | corosio (epoll) | 7.00 µs | 4.74% | +11.4% +| `concurrent_lockless/1` | asio (coroutines, epoll) | 8.29 µs | 3.82% | -5.0% +| `concurrent_lockless/1` | asio (callbacks, epoll) | 7.90 µs | 1.09% | baseline (epoll) +| `concurrent_lockless/16` | corosio (epoll) | 108.75 µs | 0.34% | -7.5% +| `concurrent_lockless/16` | asio (coroutines, epoll) | 107.97 µs | 0.77% | -6.7% +| `concurrent_lockless/16` | asio (callbacks, epoll) | 101.16 µs | 1.07% | baseline (epoll) +| `concurrent_lockless/4` | corosio (epoll) | 27.60 µs | 1.36% | -4.9% +| `concurrent_lockless/4` | asio (coroutines, epoll) | 28.47 µs | 0.49% | -8.2% +| `concurrent_lockless/4` | asio (callbacks, epoll) | 26.31 µs | 1.67% | baseline (epoll) +| `pingpong/1` | corosio (epoll) | 7.07 µs | 3.51% | +13.7% +| `pingpong/1` | asio (coroutines, epoll) | 8.56 µs | 3.08% | -4.5% +| `pingpong/1` | asio (callbacks, epoll) | 8.19 µs | 3.62% | baseline (epoll) +| `pingpong/1024` | corosio (epoll) | 7.36 µs | 4.39% | +17.7% +| `pingpong/1024` | asio (coroutines, epoll) | 9.50 µs | 3.30% | -6.1% +| `pingpong/1024` | asio (callbacks, epoll) | 8.95 µs | 3.55% | baseline (epoll) +| `pingpong/64` | corosio (epoll) | 7.11 µs | 0.18% | +13.5% +| `pingpong/64` | asio (coroutines, epoll) | 8.61 µs | 3.70% | -4.8% +| `pingpong/64` | asio (callbacks, epoll) | 8.22 µs | 0.84% | baseline (epoll) +| `pingpong_lockless/1` | corosio (epoll) | 6.96 µs | 5.13% | +11.7% +| `pingpong_lockless/1` | asio (coroutines, epoll) | 8.22 µs | 3.88% | -4.3% +| `pingpong_lockless/1` | asio (callbacks, epoll) | 7.88 µs | 4.43% | baseline (epoll) +| `pingpong_lockless/1024` | corosio (epoll) | 7.85 µs | 2.88% | +10.9% +| `pingpong_lockless/1024` | asio (coroutines, epoll) | 9.19 µs | 2.99% | -4.3% +| `pingpong_lockless/1024` | asio (callbacks, epoll) | 8.81 µs | 2.58% | baseline (epoll) +| `pingpong_lockless/64` | corosio (epoll) | 7.51 µs | 4.31% | +5.0% +| `pingpong_lockless/64` | asio (coroutines, epoll) | 8.28 µs | 4.12% | -4.7% +| `pingpong_lockless/64` | asio (callbacks, epoll) | 7.90 µs | 4.42% | baseline (epoll) +|=== +==== + +-- + +++++ +
+++++ + +=== `local_socket_throughput` + +Sustained byte throughput of a Unix domain stream socket pair under continuous streaming across chunk sizes and directions. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-local_socket_throughput-uring-g1.svg[local socket throughput comparison (io_uring), part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (io_uring) | 302.6 MB/s | 1.03% | +10.3% +| `bidirectional/1024` | asio (coroutines, io_uring) | 260.4 MB/s | 5.00% | -5.1% +| `bidirectional/1024` | asio (callbacks, io_uring) | 274.3 MB/s | 3.34% | baseline (io_uring) +| `bidirectional/1048576` | corosio (io_uring) | 3.31 GB/s | 1.57% | -4.8% +| `bidirectional/1048576` | asio (coroutines, io_uring) | 3.53 GB/s | 2.27% | +1.5% +| `bidirectional/1048576` | asio (callbacks, io_uring) | 3.47 GB/s | 2.38% | baseline (io_uring) +| `bidirectional/16384` | corosio (io_uring) | 2.42 GB/s | 1.24% | -16.1% +| `bidirectional/16384` | asio (coroutines, io_uring) | 2.88 GB/s | 2.69% | -0.2% +| `bidirectional/16384` | asio (callbacks, io_uring) | 2.88 GB/s | 1.71% | baseline (io_uring) +| `bidirectional/262144` | corosio (io_uring) | 3.30 GB/s | 1.11% | -5.6% +| `bidirectional/262144` | asio (coroutines, io_uring) | 3.52 GB/s | 1.94% | +0.5% +| `bidirectional/262144` | asio (callbacks, io_uring) | 3.50 GB/s | 2.59% | baseline (io_uring) +| `bidirectional/4096` | corosio (io_uring) | 996.0 MB/s | 1.15% | +7.0% +| `bidirectional/4096` | asio (coroutines, io_uring) | 900.3 MB/s | 2.84% | -3.3% +| `bidirectional/4096` | asio (callbacks, io_uring) | 930.6 MB/s | 2.89% | baseline (io_uring) +| `bidirectional/65536` | corosio (io_uring) | 3.02 GB/s | 2.29% | -20.0% +| `bidirectional/65536` | asio (coroutines, io_uring) | 3.72 GB/s | 4.19% | -1.6% +| `bidirectional/65536` | asio (callbacks, io_uring) | 3.77 GB/s | 2.09% | baseline (io_uring) +| `bidirectional_lockless/1024` | corosio (io_uring) | 303.0 MB/s | 0.32% | +4.9% +| `bidirectional_lockless/1024` | asio (coroutines, io_uring) | 295.7 MB/s | 3.42% | +2.4% +| `bidirectional_lockless/1024` | asio (callbacks, io_uring) | 288.8 MB/s | 4.79% | baseline (io_uring) +| `bidirectional_lockless/1048576` | corosio (io_uring) | 3.41 GB/s | 2.52% | -4.2% +| `bidirectional_lockless/1048576` | asio (coroutines, io_uring) | 3.53 GB/s | 1.21% | -0.9% +| `bidirectional_lockless/1048576` | asio (callbacks, io_uring) | 3.56 GB/s | 2.31% | baseline (io_uring) +| `bidirectional_lockless/16384` | corosio (io_uring) | 2.64 GB/s | 1.44% | -9.9% +| `bidirectional_lockless/16384` | asio (coroutines, io_uring) | 2.93 GB/s | 1.86% | +0.2% +| `bidirectional_lockless/16384` | asio (callbacks, io_uring) | 2.93 GB/s | 2.88% | baseline (io_uring) +| `bidirectional_lockless/262144` | corosio (io_uring) | 3.38 GB/s | 1.45% | -5.5% +| `bidirectional_lockless/262144` | asio (coroutines, io_uring) | 3.57 GB/s | 2.53% | -0.3% +| `bidirectional_lockless/262144` | asio (callbacks, io_uring) | 3.58 GB/s | 2.74% | baseline (io_uring) +| `bidirectional_lockless/4096` | corosio (io_uring) | 999.9 MB/s | 0.56% | +11.6% +| `bidirectional_lockless/4096` | asio (coroutines, io_uring) | 925.8 MB/s | 2.49% | +3.3% +| `bidirectional_lockless/4096` | asio (callbacks, io_uring) | 895.8 MB/s | 4.44% | baseline (io_uring) +| `bidirectional_lockless/65536` | corosio (io_uring) | 3.21 GB/s | 1.77% | -14.8% +| `bidirectional_lockless/65536` | asio (coroutines, io_uring) | 3.76 GB/s | 4.44% | -0.2% +| `bidirectional_lockless/65536` | asio (callbacks, io_uring) | 3.77 GB/s | 2.17% | baseline (io_uring) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-local_socket_throughput-uring-g2.svg[local socket throughput comparison (io_uring), part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (io_uring) | 303.5 MB/s | 0.82% | +15.0% +| `unidirectional/1024` | asio (coroutines, io_uring) | 240.0 MB/s | 3.83% | -9.0% +| `unidirectional/1024` | asio (callbacks, io_uring) | 263.8 MB/s | 3.17% | baseline (io_uring) +| `unidirectional/1048576` | corosio (io_uring) | 3.42 GB/s | 3.01% | -3.5% +| `unidirectional/1048576` | asio (coroutines, io_uring) | 3.54 GB/s | 2.17% | -0.1% +| `unidirectional/1048576` | asio (callbacks, io_uring) | 3.54 GB/s | 2.60% | baseline (io_uring) +| `unidirectional/16384` | corosio (io_uring) | 2.67 GB/s | 2.34% | -4.0% +| `unidirectional/16384` | asio (coroutines, io_uring) | 2.73 GB/s | 2.13% | -1.7% +| `unidirectional/16384` | asio (callbacks, io_uring) | 2.78 GB/s | 3.80% | baseline (io_uring) +| `unidirectional/262144` | corosio (io_uring) | 3.46 GB/s | 1.32% | -1.9% +| `unidirectional/262144` | asio (coroutines, io_uring) | 3.55 GB/s | 1.01% | +0.7% +| `unidirectional/262144` | asio (callbacks, io_uring) | 3.53 GB/s | 2.29% | baseline (io_uring) +| `unidirectional/4096` | corosio (io_uring) | 1.00 GB/s | 1.12% | +14.9% +| `unidirectional/4096` | asio (coroutines, io_uring) | 845.2 MB/s | 3.36% | -3.0% +| `unidirectional/4096` | asio (callbacks, io_uring) | 870.9 MB/s | 0.72% | baseline (io_uring) +| `unidirectional/65536` | corosio (io_uring) | 3.32 GB/s | 3.03% | -15.9% +| `unidirectional/65536` | asio (coroutines, io_uring) | 3.95 GB/s | 5.85% | -0.1% +| `unidirectional/65536` | asio (callbacks, io_uring) | 3.95 GB/s | 3.95% | baseline (io_uring) +| `unidirectional_lockless/1024` | corosio (io_uring) | 304.2 MB/s | 0.88% | +16.8% +| `unidirectional_lockless/1024` | asio (coroutines, io_uring) | 269.5 MB/s | 4.38% | +3.4% +| `unidirectional_lockless/1024` | asio (callbacks, io_uring) | 260.6 MB/s | 4.69% | baseline (io_uring) +| `unidirectional_lockless/1048576` | corosio (io_uring) | 3.41 GB/s | 2.81% | -4.2% +| `unidirectional_lockless/1048576` | asio (coroutines, io_uring) | 3.56 GB/s | 1.41% | +0.1% +| `unidirectional_lockless/1048576` | asio (callbacks, io_uring) | 3.56 GB/s | 2.67% | baseline (io_uring) +| `unidirectional_lockless/16384` | corosio (io_uring) | 2.65 GB/s | 1.33% | -5.1% +| `unidirectional_lockless/16384` | asio (coroutines, io_uring) | 2.83 GB/s | 0.69% | +1.4% +| `unidirectional_lockless/16384` | asio (callbacks, io_uring) | 2.79 GB/s | 2.48% | baseline (io_uring) +| `unidirectional_lockless/262144` | corosio (io_uring) | 3.45 GB/s | 1.24% | -3.2% +| `unidirectional_lockless/262144` | asio (coroutines, io_uring) | 3.56 GB/s | 1.25% | +0.0% +| `unidirectional_lockless/262144` | asio (callbacks, io_uring) | 3.56 GB/s | 2.67% | baseline (io_uring) +| `unidirectional_lockless/4096` | corosio (io_uring) | 1.00 GB/s | 0.67% | +11.3% +| `unidirectional_lockless/4096` | asio (coroutines, io_uring) | 872.1 MB/s | 2.14% | -3.0% +| `unidirectional_lockless/4096` | asio (callbacks, io_uring) | 899.2 MB/s | 3.15% | baseline (io_uring) +| `unidirectional_lockless/65536` | corosio (io_uring) | 3.28 GB/s | 1.64% | -17.7% +| `unidirectional_lockless/65536` | asio (coroutines, io_uring) | 3.99 GB/s | 5.86% | +0.2% +| `unidirectional_lockless/65536` | asio (callbacks, io_uring) | 3.98 GB/s | 4.49% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-local_socket_throughput-epoll-g1.svg[local socket throughput comparison (epoll), part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (epoll) | 304.3 MB/s | 0.78% | +3.2% +| `bidirectional/1024` | asio (coroutines, epoll) | 263.6 MB/s | 4.34% | -10.6% +| `bidirectional/1024` | asio (callbacks, epoll) | 294.8 MB/s | 2.68% | baseline (epoll) +| `bidirectional/1048576` | corosio (epoll) | 3.13 GB/s | 1.70% | -5.0% +| `bidirectional/1048576` | asio (coroutines, epoll) | 3.20 GB/s | 2.51% | -2.8% +| `bidirectional/1048576` | asio (callbacks, epoll) | 3.30 GB/s | 3.22% | baseline (epoll) +| `bidirectional/16384` | corosio (epoll) | 3.03 GB/s | 1.53% | +1.7% +| `bidirectional/16384` | asio (coroutines, epoll) | 2.83 GB/s | 2.51% | -4.9% +| `bidirectional/16384` | asio (callbacks, epoll) | 2.98 GB/s | 2.72% | baseline (epoll) +| `bidirectional/262144` | corosio (epoll) | 3.13 GB/s | 1.23% | -4.5% +| `bidirectional/262144` | asio (coroutines, epoll) | 3.27 GB/s | 2.17% | -0.3% +| `bidirectional/262144` | asio (callbacks, epoll) | 3.28 GB/s | 4.57% | baseline (epoll) +| `bidirectional/4096` | corosio (epoll) | 1.00 GB/s | 0.46% | +7.6% +| `bidirectional/4096` | asio (coroutines, epoll) | 888.2 MB/s | 2.87% | -4.7% +| `bidirectional/4096` | asio (callbacks, epoll) | 932.4 MB/s | 3.54% | baseline (epoll) +| `bidirectional/65536` | corosio (epoll) | 2.98 GB/s | 1.58% | -18.0% +| `bidirectional/65536` | asio (coroutines, epoll) | 3.60 GB/s | 1.68% | -1.0% +| `bidirectional/65536` | asio (callbacks, epoll) | 3.64 GB/s | 2.98% | baseline (epoll) +| `bidirectional_lockless/1024` | corosio (epoll) | 305.2 MB/s | 1.08% | +7.3% +| `bidirectional_lockless/1024` | asio (coroutines, epoll) | 265.0 MB/s | 5.06% | -6.8% +| `bidirectional_lockless/1024` | asio (callbacks, epoll) | 284.4 MB/s | 4.56% | baseline (epoll) +| `bidirectional_lockless/1048576` | corosio (epoll) | 3.16 GB/s | 0.89% | -3.4% +| `bidirectional_lockless/1048576` | asio (coroutines, epoll) | 3.22 GB/s | 2.67% | -1.5% +| `bidirectional_lockless/1048576` | asio (callbacks, epoll) | 3.27 GB/s | 4.06% | baseline (epoll) +| `bidirectional_lockless/16384` | corosio (epoll) | 3.11 GB/s | 2.10% | +1.9% +| `bidirectional_lockless/16384` | asio (coroutines, epoll) | 2.90 GB/s | 2.33% | -5.2% +| `bidirectional_lockless/16384` | asio (callbacks, epoll) | 3.06 GB/s | 1.54% | baseline (epoll) +| `bidirectional_lockless/262144` | corosio (epoll) | 3.12 GB/s | 2.30% | -4.5% +| `bidirectional_lockless/262144` | asio (coroutines, epoll) | 3.28 GB/s | 2.37% | +0.4% +| `bidirectional_lockless/262144` | asio (callbacks, epoll) | 3.27 GB/s | 4.17% | baseline (epoll) +| `bidirectional_lockless/4096` | corosio (epoll) | 1.01 GB/s | 0.84% | +6.6% +| `bidirectional_lockless/4096` | asio (coroutines, epoll) | 910.5 MB/s | 3.20% | -4.1% +| `bidirectional_lockless/4096` | asio (callbacks, epoll) | 949.8 MB/s | 3.41% | baseline (epoll) +| `bidirectional_lockless/65536` | corosio (epoll) | 3.00 GB/s | 1.86% | -19.2% +| `bidirectional_lockless/65536` | asio (coroutines, epoll) | 3.61 GB/s | 2.91% | -2.8% +| `bidirectional_lockless/65536` | asio (callbacks, epoll) | 3.72 GB/s | 3.56% | baseline (epoll) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-local_socket_throughput-epoll-g2.svg[local socket throughput comparison (epoll), part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (epoll) | 298.4 MB/s | 0.96% | +16.1% +| `unidirectional/1024` | asio (coroutines, epoll) | 240.9 MB/s | 3.83% | -6.3% +| `unidirectional/1024` | asio (callbacks, epoll) | 257.0 MB/s | 4.28% | baseline (epoll) +| `unidirectional/1048576` | corosio (epoll) | 3.25 GB/s | 1.31% | -6.3% +| `unidirectional/1048576` | asio (coroutines, epoll) | 3.39 GB/s | 2.40% | -2.3% +| `unidirectional/1048576` | asio (callbacks, epoll) | 3.47 GB/s | 3.83% | baseline (epoll) +| `unidirectional/16384` | corosio (epoll) | 3.08 GB/s | 1.51% | +7.9% +| `unidirectional/16384` | asio (coroutines, epoll) | 2.76 GB/s | 2.62% | -3.4% +| `unidirectional/16384` | asio (callbacks, epoll) | 2.86 GB/s | 1.81% | baseline (epoll) +| `unidirectional/262144` | corosio (epoll) | 3.27 GB/s | 1.66% | -5.6% +| `unidirectional/262144` | asio (coroutines, epoll) | 3.37 GB/s | 1.67% | -2.7% +| `unidirectional/262144` | asio (callbacks, epoll) | 3.46 GB/s | 3.93% | baseline (epoll) +| `unidirectional/4096` | corosio (epoll) | 978.7 MB/s | 1.06% | +11.0% +| `unidirectional/4096` | asio (coroutines, epoll) | 842.7 MB/s | 2.71% | -4.4% +| `unidirectional/4096` | asio (callbacks, epoll) | 881.8 MB/s | 0.49% | baseline (epoll) +| `unidirectional/65536` | corosio (epoll) | 3.29 GB/s | 2.35% | -13.9% +| `unidirectional/65536` | asio (coroutines, epoll) | 3.83 GB/s | 2.77% | +0.2% +| `unidirectional/65536` | asio (callbacks, epoll) | 3.82 GB/s | 2.69% | baseline (epoll) +| `unidirectional_lockless/1024` | corosio (epoll) | 299.0 MB/s | 0.71% | +12.9% +| `unidirectional_lockless/1024` | asio (coroutines, epoll) | 250.0 MB/s | 4.40% | -5.6% +| `unidirectional_lockless/1024` | asio (callbacks, epoll) | 264.9 MB/s | 4.61% | baseline (epoll) +| `unidirectional_lockless/1048576` | corosio (epoll) | 3.28 GB/s | 1.05% | -4.1% +| `unidirectional_lockless/1048576` | asio (coroutines, epoll) | 3.36 GB/s | 1.57% | -1.6% +| `unidirectional_lockless/1048576` | asio (callbacks, epoll) | 3.42 GB/s | 3.36% | baseline (epoll) +| `unidirectional_lockless/16384` | corosio (epoll) | 3.09 GB/s | 3.18% | +5.8% +| `unidirectional_lockless/16384` | asio (coroutines, epoll) | 2.81 GB/s | 1.94% | -3.8% +| `unidirectional_lockless/16384` | asio (callbacks, epoll) | 2.92 GB/s | 2.13% | baseline (epoll) +| `unidirectional_lockless/262144` | corosio (epoll) | 3.28 GB/s | 1.37% | -5.9% +| `unidirectional_lockless/262144` | asio (coroutines, epoll) | 3.34 GB/s | 3.68% | -4.2% +| `unidirectional_lockless/262144` | asio (callbacks, epoll) | 3.48 GB/s | 4.09% | baseline (epoll) +| `unidirectional_lockless/4096` | corosio (epoll) | 988.0 MB/s | 0.93% | +13.2% +| `unidirectional_lockless/4096` | asio (coroutines, epoll) | 869.2 MB/s | 2.52% | -0.4% +| `unidirectional_lockless/4096` | asio (callbacks, epoll) | 873.0 MB/s | 2.70% | baseline (epoll) +| `unidirectional_lockless/65536` | corosio (epoll) | 3.29 GB/s | 4.33% | -14.8% +| `unidirectional_lockless/65536` | asio (coroutines, epoll) | 3.92 GB/s | 1.70% | +1.4% +| `unidirectional_lockless/65536` | asio (callbacks, epoll) | 3.86 GB/s | 3.97% | baseline (epoll) +|=== +==== + +-- + +++++ +
+++++ + +=== `socket_latency` + +Round-trip latency of a TCP loopback connection. Each sample is one full round trip (request out, reply back) across message sizes and concurrent pair counts. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-socket_latency-uring.svg[socket latency comparison (io_uring),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (io_uring) | 14.57 µs | 0.98% | +13.3% +| `concurrent/1` | asio (coroutines, io_uring) | 17.13 µs | 0.50% | -2.0% +| `concurrent/1` | asio (callbacks, io_uring) | 16.81 µs | 1.22% | baseline (io_uring) +| `concurrent/16` | corosio (io_uring) | 236.38 µs | 0.79% | -3.8% +| `concurrent/16` | asio (coroutines, io_uring) | 232.68 µs | 0.92% | -2.2% +| `concurrent/16` | asio (callbacks, io_uring) | 227.64 µs | 0.96% | baseline (io_uring) +| `concurrent/4` | corosio (io_uring) | 58.57 µs | 0.71% | +0.5% +| `concurrent/4` | asio (coroutines, io_uring) | 60.22 µs | 0.94% | -2.3% +| `concurrent/4` | asio (callbacks, io_uring) | 58.84 µs | 0.87% | baseline (io_uring) +| `concurrent_lockless/1` | corosio (io_uring) | 14.58 µs | 1.06% | +10.7% +| `concurrent_lockless/1` | asio (coroutines, io_uring) | 16.70 µs | 1.06% | -2.3% +| `concurrent_lockless/1` | asio (callbacks, io_uring) | 16.33 µs | 0.86% | baseline (io_uring) +| `concurrent_lockless/16` | corosio (io_uring) | 235.72 µs | 1.04% | -5.0% +| `concurrent_lockless/16` | asio (coroutines, io_uring) | 228.57 µs | 1.04% | -1.8% +| `concurrent_lockless/16` | asio (callbacks, io_uring) | 224.43 µs | 1.01% | baseline (io_uring) +| `concurrent_lockless/4` | corosio (io_uring) | 58.50 µs | 1.02% | -1.2% +| `concurrent_lockless/4` | asio (coroutines, io_uring) | 59.27 µs | 1.28% | -2.5% +| `concurrent_lockless/4` | asio (callbacks, io_uring) | 57.81 µs | 1.11% | baseline (io_uring) +| `pingpong/1` | corosio (io_uring) | 14.52 µs | 1.27% | +13.3% +| `pingpong/1` | asio (coroutines, io_uring) | 17.04 µs | 1.15% | -1.7% +| `pingpong/1` | asio (callbacks, io_uring) | 16.75 µs | 0.11% | baseline (io_uring) +| `pingpong/1024` | corosio (io_uring) | 14.81 µs | 1.36% | +13.3% +| `pingpong/1024` | asio (coroutines, io_uring) | 17.46 µs | 0.82% | -2.2% +| `pingpong/1024` | asio (callbacks, io_uring) | 17.09 µs | 0.65% | baseline (io_uring) +| `pingpong/64` | corosio (io_uring) | 14.59 µs | 1.34% | +13.2% +| `pingpong/64` | asio (coroutines, io_uring) | 17.14 µs | 0.86% | -1.9% +| `pingpong/64` | asio (callbacks, io_uring) | 16.81 µs | 0.26% | baseline (io_uring) +| `pingpong_lockless/1` | corosio (io_uring) | 14.49 µs | 1.33% | +10.8% +| `pingpong_lockless/1` | asio (coroutines, io_uring) | 16.64 µs | 1.06% | -2.5% +| `pingpong_lockless/1` | asio (callbacks, io_uring) | 16.24 µs | 1.05% | baseline (io_uring) +| `pingpong_lockless/1024` | corosio (io_uring) | 14.81 µs | 1.19% | +10.8% +| `pingpong_lockless/1024` | asio (coroutines, io_uring) | 17.01 µs | 0.27% | -2.5% +| `pingpong_lockless/1024` | asio (callbacks, io_uring) | 16.59 µs | 1.12% | baseline (io_uring) +| `pingpong_lockless/64` | corosio (io_uring) | 14.56 µs | 1.35% | +10.6% +| `pingpong_lockless/64` | asio (coroutines, io_uring) | 16.73 µs | 0.21% | -2.7% +| `pingpong_lockless/64` | asio (callbacks, io_uring) | 16.29 µs | 0.97% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-socket_latency-epoll.svg[socket latency comparison (epoll),role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (epoll) | 14.80 µs | 0.88% | +7.1% +| `concurrent/1` | asio (coroutines, epoll) | 16.32 µs | 0.92% | -2.4% +| `concurrent/1` | asio (callbacks, epoll) | 15.94 µs | 0.61% | baseline (epoll) +| `concurrent/16` | corosio (epoll) | 232.79 µs | 0.98% | -1.4% +| `concurrent/16` | asio (coroutines, epoll) | 236.30 µs | 0.99% | -2.9% +| `concurrent/16` | asio (callbacks, epoll) | 229.53 µs | 0.71% | baseline (epoll) +| `concurrent/4` | corosio (epoll) | 58.22 µs | 0.88% | +0.4% +| `concurrent/4` | asio (coroutines, epoll) | 60.27 µs | 1.02% | -3.1% +| `concurrent/4` | asio (callbacks, epoll) | 58.48 µs | 0.79% | baseline (epoll) +| `concurrent_lockless/1` | corosio (epoll) | 14.69 µs | 1.26% | +6.0% +| `concurrent_lockless/1` | asio (coroutines, epoll) | 16.00 µs | 0.81% | -2.4% +| `concurrent_lockless/1` | asio (callbacks, epoll) | 15.63 µs | 0.97% | baseline (epoll) +| `concurrent_lockless/16` | corosio (epoll) | 231.82 µs | 1.01% | -3.3% +| `concurrent_lockless/16` | asio (coroutines, epoll) | 231.30 µs | 1.07% | -3.1% +| `concurrent_lockless/16` | asio (callbacks, epoll) | 224.45 µs | 0.91% | baseline (epoll) +| `concurrent_lockless/4` | corosio (epoll) | 57.92 µs | 1.30% | -1.3% +| `concurrent_lockless/4` | asio (coroutines, epoll) | 59.00 µs | 0.95% | -3.1% +| `concurrent_lockless/4` | asio (callbacks, epoll) | 57.20 µs | 1.03% | baseline (epoll) +| `pingpong/1` | corosio (epoll) | 14.74 µs | 1.03% | +7.2% +| `pingpong/1` | asio (coroutines, epoll) | 16.29 µs | 1.32% | -2.5% +| `pingpong/1` | asio (callbacks, epoll) | 15.88 µs | 1.06% | baseline (epoll) +| `pingpong/1024` | corosio (epoll) | 15.04 µs | 1.18% | +7.3% +| `pingpong/1024` | asio (coroutines, epoll) | 16.56 µs | 0.99% | -2.1% +| `pingpong/1024` | asio (callbacks, epoll) | 16.22 µs | 0.98% | baseline (epoll) +| `pingpong/64` | corosio (epoll) | 14.78 µs | 1.17% | +7.3% +| `pingpong/64` | asio (coroutines, epoll) | 16.35 µs | 1.26% | -2.6% +| `pingpong/64` | asio (callbacks, epoll) | 15.94 µs | 0.90% | baseline (epoll) +| `pingpong_lockless/1` | corosio (epoll) | 14.62 µs | 0.99% | +5.8% +| `pingpong_lockless/1` | asio (coroutines, epoll) | 15.92 µs | 0.79% | -2.5% +| `pingpong_lockless/1` | asio (callbacks, epoll) | 15.52 µs | 0.86% | baseline (epoll) +| `pingpong_lockless/1024` | corosio (epoll) | 14.96 µs | 0.95% | +5.8% +| `pingpong_lockless/1024` | asio (coroutines, epoll) | 16.26 µs | 0.90% | -2.4% +| `pingpong_lockless/1024` | asio (callbacks, epoll) | 15.89 µs | 0.12% | baseline (epoll) +| `pingpong_lockless/64` | corosio (epoll) | 14.71 µs | 0.98% | +5.8% +| `pingpong_lockless/64` | asio (coroutines, epoll) | 16.01 µs | 0.66% | -2.5% +| `pingpong_lockless/64` | asio (callbacks, epoll) | 15.61 µs | 0.18% | baseline (epoll) +|=== +==== + +-- + +++++ +
+++++ + +=== `socket_throughput` + +Sustained byte throughput of a TCP loopback connection under continuous streaming, varying chunk size, direction, and concurrency. + +**io_uring** + +[.bch-card] +-- +image::bench/linux-socket_throughput-uring-g1.svg[socket throughput comparison (io_uring), part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (io_uring) | 140.6 MB/s | 1.04% | +2.9% +| `bidirectional/1024` | asio (coroutines, io_uring) | 134.2 MB/s | 1.24% | -1.8% +| `bidirectional/1024` | asio (callbacks, io_uring) | 136.6 MB/s | 0.45% | baseline (io_uring) +| `bidirectional/1048576` | corosio (io_uring) | 3.26 GB/s | 1.47% | -0.7% +| `bidirectional/1048576` | asio (coroutines, io_uring) | 3.25 GB/s | 0.88% | -0.8% +| `bidirectional/1048576` | asio (callbacks, io_uring) | 3.28 GB/s | 0.86% | baseline (io_uring) +| `bidirectional/16384` | corosio (io_uring) | 1.54 GB/s | 0.62% | -6.2% +| `bidirectional/16384` | asio (coroutines, io_uring) | 1.56 GB/s | 0.47% | -5.1% +| `bidirectional/16384` | asio (callbacks, io_uring) | 1.64 GB/s | 1.87% | baseline (io_uring) +| `bidirectional/262144` | corosio (io_uring) | 3.21 GB/s | 1.11% | +10.8% +| `bidirectional/262144` | asio (coroutines, io_uring) | 2.86 GB/s | 0.72% | -1.1% +| `bidirectional/262144` | asio (callbacks, io_uring) | 2.89 GB/s | 1.61% | baseline (io_uring) +| `bidirectional/4096` | corosio (io_uring) | 494.8 MB/s | 1.09% | +0.4% +| `bidirectional/4096` | asio (coroutines, io_uring) | 485.4 MB/s | 1.25% | -1.5% +| `bidirectional/4096` | asio (callbacks, io_uring) | 492.8 MB/s | 1.39% | baseline (io_uring) +| `bidirectional/65536` | corosio (io_uring) | 2.76 GB/s | 2.00% | +5.0% +| `bidirectional/65536` | asio (coroutines, io_uring) | 2.60 GB/s | 1.95% | -1.0% +| `bidirectional/65536` | asio (callbacks, io_uring) | 2.63 GB/s | 2.11% | baseline (io_uring) +| `bidirectional_lockless/1024` | corosio (io_uring) | 140.8 MB/s | 0.18% | +0.9% +| `bidirectional_lockless/1024` | asio (coroutines, io_uring) | 136.4 MB/s | 0.13% | -2.2% +| `bidirectional_lockless/1024` | asio (callbacks, io_uring) | 139.5 MB/s | 1.34% | baseline (io_uring) +| `bidirectional_lockless/1048576` | corosio (io_uring) | 3.37 GB/s | 1.25% | +3.0% +| `bidirectional_lockless/1048576` | asio (coroutines, io_uring) | 3.28 GB/s | 1.49% | +0.2% +| `bidirectional_lockless/1048576` | asio (callbacks, io_uring) | 3.27 GB/s | 1.11% | baseline (io_uring) +| `bidirectional_lockless/16384` | corosio (io_uring) | 1.54 GB/s | 1.96% | -8.0% +| `bidirectional_lockless/16384` | asio (coroutines, io_uring) | 1.66 GB/s | 1.10% | -0.5% +| `bidirectional_lockless/16384` | asio (callbacks, io_uring) | 1.67 GB/s | 2.44% | baseline (io_uring) +| `bidirectional_lockless/262144` | corosio (io_uring) | 3.25 GB/s | 1.02% | +13.0% +| `bidirectional_lockless/262144` | asio (coroutines, io_uring) | 2.90 GB/s | 0.99% | +0.8% +| `bidirectional_lockless/262144` | asio (callbacks, io_uring) | 2.88 GB/s | 0.46% | baseline (io_uring) +| `bidirectional_lockless/4096` | corosio (io_uring) | 494.3 MB/s | 0.58% | -1.9% +| `bidirectional_lockless/4096` | asio (coroutines, io_uring) | 494.6 MB/s | 0.28% | -1.8% +| `bidirectional_lockless/4096` | asio (callbacks, io_uring) | 503.6 MB/s | 1.55% | baseline (io_uring) +| `bidirectional_lockless/65536` | corosio (io_uring) | 2.78 GB/s | 0.90% | +5.1% +| `bidirectional_lockless/65536` | asio (coroutines, io_uring) | 2.61 GB/s | 1.69% | -1.1% +| `bidirectional_lockless/65536` | asio (callbacks, io_uring) | 2.64 GB/s | 2.18% | baseline (io_uring) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-socket_throughput-uring-g2.svg[socket throughput comparison (io_uring), part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `multithread/2` | corosio (io_uring) | 3.67 GB/s | 1.05% | +123.3% +| `multithread/2` | asio (coroutines, io_uring) | 1.75 GB/s | 7.63% | +6.8% +| `multithread/2` | asio (callbacks, io_uring) | 1.64 GB/s | 7.69% | baseline (io_uring) +| `multithread/4` | corosio (io_uring) | 5.08 GB/s | 0.33% | +206.1% +| `multithread/4` | asio (coroutines, io_uring) | 1.64 GB/s | 0.65% | -1.2% +| `multithread/4` | asio (callbacks, io_uring) | 1.66 GB/s | 0.65% | baseline (io_uring) +| `multithread/8` | corosio (io_uring) | 5.13 GB/s | 0.17% | +214.9% +| `multithread/8` | asio (coroutines, io_uring) | 1.61 GB/s | 0.94% | -1.0% +| `multithread/8` | asio (callbacks, io_uring) | 1.63 GB/s | 0.60% | baseline (io_uring) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-socket_throughput-uring-g3.svg[socket throughput comparison (io_uring), part 3,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (io_uring) | 139.2 MB/s | 0.31% | +51.1% +| `unidirectional/1024` | asio (coroutines, io_uring) | 90.70 MB/s | 0.50% | -1.6% +| `unidirectional/1024` | asio (callbacks, io_uring) | 92.14 MB/s | 1.18% | baseline (io_uring) +| `unidirectional/1048576` | corosio (io_uring) | 3.41 GB/s | 1.66% | +3.5% +| `unidirectional/1048576` | asio (coroutines, io_uring) | 3.31 GB/s | 2.10% | +0.3% +| `unidirectional/1048576` | asio (callbacks, io_uring) | 3.30 GB/s | 0.48% | baseline (io_uring) +| `unidirectional/16384` | corosio (io_uring) | 1.54 GB/s | 3.75% | +16.9% +| `unidirectional/16384` | asio (coroutines, io_uring) | 1.32 GB/s | 1.75% | -0.2% +| `unidirectional/16384` | asio (callbacks, io_uring) | 1.32 GB/s | 1.06% | baseline (io_uring) +| `unidirectional/262144` | corosio (io_uring) | 3.23 GB/s | 1.54% | +6.1% +| `unidirectional/262144` | asio (coroutines, io_uring) | 3.07 GB/s | 1.19% | +1.0% +| `unidirectional/262144` | asio (callbacks, io_uring) | 3.04 GB/s | 0.67% | baseline (io_uring) +| `unidirectional/4096` | corosio (io_uring) | 491.5 MB/s | 1.27% | +36.5% +| `unidirectional/4096` | asio (coroutines, io_uring) | 354.4 MB/s | 0.83% | -1.6% +| `unidirectional/4096` | asio (callbacks, io_uring) | 360.0 MB/s | 0.52% | baseline (io_uring) +| `unidirectional/65536` | corosio (io_uring) | 2.72 GB/s | 0.87% | +3.1% +| `unidirectional/65536` | asio (coroutines, io_uring) | 2.64 GB/s | 2.73% | -0.2% +| `unidirectional/65536` | asio (callbacks, io_uring) | 2.64 GB/s | 2.12% | baseline (io_uring) +| `unidirectional_lockless/1024` | corosio (io_uring) | 139.7 MB/s | 0.91% | +49.0% +| `unidirectional_lockless/1024` | asio (coroutines, io_uring) | 92.09 MB/s | 0.45% | -1.8% +| `unidirectional_lockless/1024` | asio (callbacks, io_uring) | 93.74 MB/s | 1.32% | baseline (io_uring) +| `unidirectional_lockless/1048576` | corosio (io_uring) | 3.34 GB/s | 1.53% | -4.7% +| `unidirectional_lockless/1048576` | asio (coroutines, io_uring) | 3.52 GB/s | 0.43% | +0.5% +| `unidirectional_lockless/1048576` | asio (callbacks, io_uring) | 3.51 GB/s | 0.76% | baseline (io_uring) +| `unidirectional_lockless/16384` | corosio (io_uring) | 1.62 GB/s | 2.27% | +21.0% +| `unidirectional_lockless/16384` | asio (coroutines, io_uring) | 1.33 GB/s | 1.46% | -0.6% +| `unidirectional_lockless/16384` | asio (callbacks, io_uring) | 1.34 GB/s | 2.07% | baseline (io_uring) +| `unidirectional_lockless/262144` | corosio (io_uring) | 3.23 GB/s | 1.21% | +6.8% +| `unidirectional_lockless/262144` | asio (coroutines, io_uring) | 3.10 GB/s | 1.13% | +2.2% +| `unidirectional_lockless/262144` | asio (callbacks, io_uring) | 3.03 GB/s | 1.03% | baseline (io_uring) +| `unidirectional_lockless/4096` | corosio (io_uring) | 494.4 MB/s | 0.99% | +35.3% +| `unidirectional_lockless/4096` | asio (coroutines, io_uring) | 361.0 MB/s | 0.29% | -1.2% +| `unidirectional_lockless/4096` | asio (callbacks, io_uring) | 365.4 MB/s | 1.55% | baseline (io_uring) +| `unidirectional_lockless/65536` | corosio (io_uring) | 2.73 GB/s | 2.34% | +1.8% +| `unidirectional_lockless/65536` | asio (coroutines, io_uring) | 2.64 GB/s | 2.13% | -1.4% +| `unidirectional_lockless/65536` | asio (callbacks, io_uring) | 2.68 GB/s | 2.22% | baseline (io_uring) +|=== +==== + +-- + +**epoll** + +[.bch-card] +-- +image::bench/linux-socket_throughput-epoll-g1.svg[socket throughput comparison (epoll), part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (epoll) | 134.2 MB/s | 0.37% | -3.7% +| `bidirectional/1024` | asio (coroutines, epoll) | 134.9 MB/s | 0.79% | -3.2% +| `bidirectional/1024` | asio (callbacks, epoll) | 139.4 MB/s | 0.93% | baseline (epoll) +| `bidirectional/1048576` | corosio (epoll) | 3.24 GB/s | 2.03% | -1.1% +| `bidirectional/1048576` | asio (coroutines, epoll) | 3.26 GB/s | 1.14% | -0.7% +| `bidirectional/1048576` | asio (callbacks, epoll) | 3.28 GB/s | 0.90% | baseline (epoll) +| `bidirectional/16384` | corosio (epoll) | 1.64 GB/s | 0.44% | -1.8% +| `bidirectional/16384` | asio (coroutines, epoll) | 1.63 GB/s | 1.71% | -2.4% +| `bidirectional/16384` | asio (callbacks, epoll) | 1.67 GB/s | 1.45% | baseline (epoll) +| `bidirectional/262144` | corosio (epoll) | 3.23 GB/s | 0.81% | +12.5% +| `bidirectional/262144` | asio (coroutines, epoll) | 2.92 GB/s | 2.11% | +1.7% +| `bidirectional/262144` | asio (callbacks, epoll) | 2.87 GB/s | 0.66% | baseline (epoll) +| `bidirectional/4096` | corosio (epoll) | 506.4 MB/s | 0.56% | +1.1% +| `bidirectional/4096` | asio (coroutines, epoll) | 491.0 MB/s | 1.43% | -2.0% +| `bidirectional/4096` | asio (callbacks, epoll) | 500.9 MB/s | 1.38% | baseline (epoll) +| `bidirectional/65536` | corosio (epoll) | 2.57 GB/s | 2.19% | -2.4% +| `bidirectional/65536` | asio (coroutines, epoll) | 2.59 GB/s | 1.89% | -1.4% +| `bidirectional/65536` | asio (callbacks, epoll) | 2.63 GB/s | 1.51% | baseline (epoll) +| `bidirectional_lockless/1024` | corosio (epoll) | 135.0 MB/s | 1.20% | -4.5% +| `bidirectional_lockless/1024` | asio (coroutines, epoll) | 137.6 MB/s | 1.28% | -2.7% +| `bidirectional_lockless/1024` | asio (callbacks, epoll) | 141.4 MB/s | 0.91% | baseline (epoll) +| `bidirectional_lockless/1048576` | corosio (epoll) | 3.25 GB/s | 2.77% | -2.2% +| `bidirectional_lockless/1048576` | asio (coroutines, epoll) | 3.27 GB/s | 0.87% | -1.6% +| `bidirectional_lockless/1048576` | asio (callbacks, epoll) | 3.33 GB/s | 1.88% | baseline (epoll) +| `bidirectional_lockless/16384` | corosio (epoll) | 1.65 GB/s | 0.29% | -2.4% +| `bidirectional_lockless/16384` | asio (coroutines, epoll) | 1.67 GB/s | 0.77% | -1.4% +| `bidirectional_lockless/16384` | asio (callbacks, epoll) | 1.69 GB/s | 1.75% | baseline (epoll) +| `bidirectional_lockless/262144` | corosio (epoll) | 3.23 GB/s | 1.02% | +10.0% +| `bidirectional_lockless/262144` | asio (coroutines, epoll) | 2.86 GB/s | 1.38% | -2.5% +| `bidirectional_lockless/262144` | asio (callbacks, epoll) | 2.93 GB/s | 2.33% | baseline (epoll) +| `bidirectional_lockless/4096` | corosio (epoll) | 510.0 MB/s | 1.42% | -0.2% +| `bidirectional_lockless/4096` | asio (coroutines, epoll) | 496.9 MB/s | 1.06% | -2.8% +| `bidirectional_lockless/4096` | asio (callbacks, epoll) | 511.1 MB/s | 1.13% | baseline (epoll) +| `bidirectional_lockless/65536` | corosio (epoll) | 2.52 GB/s | 3.26% | -5.2% +| `bidirectional_lockless/65536` | asio (coroutines, epoll) | 2.60 GB/s | 2.95% | -2.4% +| `bidirectional_lockless/65536` | asio (callbacks, epoll) | 2.66 GB/s | 2.19% | baseline (epoll) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-socket_throughput-epoll-g2.svg[socket throughput comparison (epoll), part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `multithread/2` | corosio (epoll) | 3.40 GB/s | 0.78% | +44.6% +| `multithread/2` | asio (coroutines, epoll) | 2.28 GB/s | 2.00% | -2.9% +| `multithread/2` | asio (callbacks, epoll) | 2.35 GB/s | 1.70% | baseline (epoll) +| `multithread/4` | corosio (epoll) | 4.87 GB/s | 0.44% | +38.0% +| `multithread/4` | asio (coroutines, epoll) | 3.53 GB/s | 0.56% | +0.1% +| `multithread/4` | asio (callbacks, epoll) | 3.53 GB/s | 0.57% | baseline (epoll) +| `multithread/8` | corosio (epoll) | 4.95 GB/s | 0.22% | +25.4% +| `multithread/8` | asio (coroutines, epoll) | 3.94 GB/s | 0.19% | -0.3% +| `multithread/8` | asio (callbacks, epoll) | 3.95 GB/s | 0.35% | baseline (epoll) +|=== +==== + +-- + +[.bch-card] +-- +image::bench/linux-socket_throughput-epoll-g3.svg[socket throughput comparison (epoll), part 3,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (epoll) | 126.1 MB/s | 1.10% | +34.6% +| `unidirectional/1024` | asio (coroutines, epoll) | 91.85 MB/s | 0.45% | -1.9% +| `unidirectional/1024` | asio (callbacks, epoll) | 93.66 MB/s | 0.88% | baseline (epoll) +| `unidirectional/1048576` | corosio (epoll) | 3.38 GB/s | 0.96% | -0.7% +| `unidirectional/1048576` | asio (coroutines, epoll) | 3.36 GB/s | 2.04% | -1.3% +| `unidirectional/1048576` | asio (callbacks, epoll) | 3.40 GB/s | 0.40% | baseline (epoll) +| `unidirectional/16384` | corosio (epoll) | 1.58 GB/s | 1.29% | +15.6% +| `unidirectional/16384` | asio (coroutines, epoll) | 1.33 GB/s | 1.39% | -3.0% +| `unidirectional/16384` | asio (callbacks, epoll) | 1.37 GB/s | 1.28% | baseline (epoll) +| `unidirectional/262144` | corosio (epoll) | 3.35 GB/s | 1.21% | +10.0% +| `unidirectional/262144` | asio (coroutines, epoll) | 3.04 GB/s | 1.49% | -0.2% +| `unidirectional/262144` | asio (callbacks, epoll) | 3.04 GB/s | 2.22% | baseline (epoll) +| `unidirectional/4096` | corosio (epoll) | 477.0 MB/s | 1.07% | +30.0% +| `unidirectional/4096` | asio (coroutines, epoll) | 358.2 MB/s | 0.83% | -2.4% +| `unidirectional/4096` | asio (callbacks, epoll) | 366.8 MB/s | 0.70% | baseline (epoll) +| `unidirectional/65536` | corosio (epoll) | 2.76 GB/s | 3.74% | +2.4% +| `unidirectional/65536` | asio (coroutines, epoll) | 2.69 GB/s | 1.14% | -0.3% +| `unidirectional/65536` | asio (callbacks, epoll) | 2.70 GB/s | 2.02% | baseline (epoll) +| `unidirectional_lockless/1024` | corosio (epoll) | 126.1 MB/s | 0.80% | +33.3% +| `unidirectional_lockless/1024` | asio (coroutines, epoll) | 93.22 MB/s | 1.39% | -1.5% +| `unidirectional_lockless/1024` | asio (callbacks, epoll) | 94.59 MB/s | 0.33% | baseline (epoll) +| `unidirectional_lockless/1048576` | corosio (epoll) | 3.35 GB/s | 1.22% | -6.3% +| `unidirectional_lockless/1048576` | asio (coroutines, epoll) | 3.55 GB/s | 2.34% | -0.8% +| `unidirectional_lockless/1048576` | asio (callbacks, epoll) | 3.57 GB/s | 1.29% | baseline (epoll) +| `unidirectional_lockless/16384` | corosio (epoll) | 1.56 GB/s | 2.38% | +13.6% +| `unidirectional_lockless/16384` | asio (coroutines, epoll) | 1.35 GB/s | 1.63% | -2.0% +| `unidirectional_lockless/16384` | asio (callbacks, epoll) | 1.37 GB/s | 1.51% | baseline (epoll) +| `unidirectional_lockless/262144` | corosio (epoll) | 3.35 GB/s | 1.59% | +7.5% +| `unidirectional_lockless/262144` | asio (coroutines, epoll) | 3.07 GB/s | 1.21% | -1.4% +| `unidirectional_lockless/262144` | asio (callbacks, epoll) | 3.11 GB/s | 1.77% | baseline (epoll) +| `unidirectional_lockless/4096` | corosio (epoll) | 476.5 MB/s | 1.12% | +28.4% +| `unidirectional_lockless/4096` | asio (coroutines, epoll) | 364.4 MB/s | 1.20% | -1.8% +| `unidirectional_lockless/4096` | asio (callbacks, epoll) | 371.0 MB/s | 0.39% | baseline (epoll) +| `unidirectional_lockless/65536` | corosio (epoll) | 2.74 GB/s | 4.50% | +0.3% +| `unidirectional_lockless/65536` | asio (coroutines, epoll) | 2.73 GB/s | 2.28% | +0.1% +| `unidirectional_lockless/65536` | asio (callbacks, epoll) | 2.73 GB/s | 2.83% | baseline (epoll) +|=== +==== + +-- + diff --git a/doc/modules/ROOT/pages/benchmarks/macos.adoc b/doc/modules/ROOT/pages/benchmarks/macos.adoc new file mode 100644 index 000000000..1b6051c73 --- /dev/null +++ b/doc/modules/ROOT/pages/benchmarks/macos.adoc @@ -0,0 +1,595 @@ += macOS Benchmarks +:page-mode: explanation +:toc: left + +_Generated 2026-10-02 from corosio `e2de06fda069` — https://github.com/cppalliance/corosio/actions/runs/37015995763[benchmark run]. This page is fully generated; do not edit by hand (see the xref:benchmarks/index.adoc[Benchmarks landing page])._ + +== Summary + +_Within noise_ means the relative difference is within twice the combined run-to-run noise (the root-sum-square of each side's CV). Differences that small are indistinguishable from measurement jitter. See xref:benchmarks/index.adoc[Methodology] for how these figures are computed. + +**68 faster · 15 within noise · 30 slower** of 113 benchmarks — median **+58.8%** vs Boost.Asio (callbacks) on the same reactor. + +image::bench/macos-summary.svg[summary,role=bch-block,opts=inline] + +== Test Environment + +[cols="1,3"] +|=== +| CPU | Apple M1 +| Cores | 8 +| RAM (GB) | 16 +| OS | Darwin 25.6.0 +| Kernel/build | Darwin Kernel Version 25.6.0: Tue Aug 18 17:48:35 PDT 2026; root:xnu-12377.161.15.700.19~2/RELEASE_ARM64_T8103 +| Compiler | Apple clang version 17.0.0 (clang-1700.6.3.2) +| CMake | `cmake version 4.4.3` +| liburing | n/a +| Boost commit | 39fa1e9a3491bd099b570935b3f3422065f91b03 +| Asio commit | a7dc25b4cb6c49a6946d86ea20664f1027203225 +| Asio reactor | kqueue +| Capy commit | a372a6b054261f29497ac0d19c2a9533f9eaad40 +| Corosio commit | e2de06fda069607f34bc957cbefcbd9ba12fddca +| Corosio branch | pr/benchmark-report +| Date (UTC) | 2026-10-02 +| Iterations | 7 +| Duration per benchmark (s) | 2.0 +|=== + +== Results + +=== `accept_churn` + +Rate of setting up and tearing down short-lived TCP connections: connect, accept, close. + +[.bch-card] +-- +image::bench/macos-accept_churn.svg[accept churn comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `burst/10` | corosio (kqueue) | 4.12K ops/s | 4.19% | -10.2% +| `burst/10` | asio (coroutines) | 3.18K ops/s | 1.54% | -30.6% +| `burst/10` | asio (callbacks) | 4.58K ops/s | 2.22% | baseline +| `burst/100` | corosio (kqueue) | 426.7 ops/s | 4.25% | -27.3% +| `burst/100` | asio (coroutines) | 343.8 ops/s | 2.96% | -41.4% +| `burst/100` | asio (callbacks) | 587.1 ops/s | 12.84% | baseline +| `burst_lockless/10` | corosio (kqueue) | 4.11K ops/s | 3.92% | -10.5% +| `burst_lockless/10` | asio (coroutines) | 3.24K ops/s | 1.91% | -29.4% +| `burst_lockless/10` | asio (callbacks) | 4.59K ops/s | 9.95% | baseline +| `burst_lockless/100` | corosio (kqueue) | 424.0 ops/s | 3.50% | -28.0% +| `burst_lockless/100` | asio (coroutines) | 347.1 ops/s | 2.79% | -41.0% +| `burst_lockless/100` | asio (callbacks) | 588.5 ops/s | 4.01% | baseline +| `concurrent/1` | corosio (kqueue) | 19.67K ops/s | 0.52% | +9.7% +| `concurrent/1` | asio (coroutines) | 17.79K ops/s | 1.75% | -0.7% +| `concurrent/1` | asio (callbacks) | 17.93K ops/s | 0.74% | baseline +| `concurrent/16` | corosio (kqueue) | 34.00K ops/s | 3.18% | -20.6% +| `concurrent/16` | asio (coroutines) | 42.53K ops/s | 2.55% | -0.7% +| `concurrent/16` | asio (callbacks) | 42.81K ops/s | 6.50% | baseline +| `concurrent/4` | corosio (kqueue) | 28.71K ops/s | 2.41% | -8.2% +| `concurrent/4` | asio (coroutines) | 30.05K ops/s | 1.48% | -3.9% +| `concurrent/4` | asio (callbacks) | 31.27K ops/s | 3.26% | baseline +| `sequential` | corosio (kqueue) | 19.68K ops/s | 0.50% | +9.1% +| `sequential` | asio (coroutines) | 17.81K ops/s | 0.97% | -1.3% +| `sequential` | asio (callbacks) | 18.05K ops/s | 0.54% | baseline +| `sequential_lockless` | corosio (kqueue) | 19.73K ops/s | 1.83% | +9.8% +| `sequential_lockless` | asio (coroutines) | 17.79K ops/s | 1.50% | -1.0% +| `sequential_lockless` | asio (callbacks) | 17.97K ops/s | 0.50% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `fan_out` + +Fan-out/fan-in coroutine coordination: a parent starts concurrent sub-requests against echo servers and awaits their completion via a shared latch. + +[.bch-card] +-- +image::bench/macos-fan_out-g1.svg[fan out comparison, part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent_parents/1` | corosio (kqueue) | 7.72K ops/s | 0.42% | -4.7% +| `concurrent_parents/1` | asio (coroutines) | 7.20K ops/s | 0.22% | -11.1% +| `concurrent_parents/1` | asio (callbacks) | 8.10K ops/s | 0.65% | baseline +| `concurrent_parents/16` | corosio (kqueue) | 8.45K ops/s | 19.25% | -7.3% +| `concurrent_parents/16` | asio (coroutines) | 6.35K ops/s | 19.09% | -30.3% +| `concurrent_parents/16` | asio (callbacks) | 9.11K ops/s | 17.43% | baseline +| `concurrent_parents/4` | corosio (kqueue) | 8.25K ops/s | 1.64% | -7.8% +| `concurrent_parents/4` | asio (coroutines) | 8.47K ops/s | 0.83% | -5.4% +| `concurrent_parents/4` | asio (callbacks) | 8.95K ops/s | 0.17% | baseline +| `concurrent_parents_lockless/1` | corosio (kqueue) | 7.76K ops/s | 0.94% | -4.8% +| `concurrent_parents_lockless/1` | asio (coroutines) | 7.25K ops/s | 0.21% | -11.1% +| `concurrent_parents_lockless/1` | asio (callbacks) | 8.16K ops/s | 0.69% | baseline +| `concurrent_parents_lockless/16` | corosio (kqueue) | 8.29K ops/s | 14.13% | -10.3% +| `concurrent_parents_lockless/16` | asio (coroutines) | 6.34K ops/s | 18.94% | -31.4% +| `concurrent_parents_lockless/16` | asio (callbacks) | 9.24K ops/s | 10.75% | baseline +| `concurrent_parents_lockless/4` | corosio (kqueue) | 8.10K ops/s | 1.89% | -10.5% +| `concurrent_parents_lockless/4` | asio (coroutines) | 8.65K ops/s | 0.83% | -4.4% +| `concurrent_parents_lockless/4` | asio (callbacks) | 9.05K ops/s | 0.31% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/macos-fan_out-g2.svg[fan out comparison, part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `fork_join/1` | corosio (kqueue) | 46.39K ops/s | 0.98% | -23.4% +| `fork_join/1` | asio (coroutines) | 29.26K ops/s | 1.31% | -51.7% +| `fork_join/1` | asio (callbacks) | 60.56K ops/s | 1.56% | baseline +| `fork_join/16` | corosio (kqueue) | 7.70K ops/s | 0.41% | -4.1% +| `fork_join/16` | asio (coroutines) | 7.21K ops/s | 0.18% | -10.2% +| `fork_join/16` | asio (callbacks) | 8.03K ops/s | 0.70% | baseline +| `fork_join/4` | corosio (kqueue) | 21.70K ops/s | 0.41% | -6.7% +| `fork_join/4` | asio (coroutines) | 16.80K ops/s | 0.33% | -27.8% +| `fork_join/4` | asio (callbacks) | 23.27K ops/s | 0.31% | baseline +| `fork_join/64` | corosio (kqueue) | 1.99K ops/s | 2.25% | -11.2% +| `fork_join/64` | asio (coroutines) | 2.08K ops/s | 0.97% | -7.4% +| `fork_join/64` | asio (callbacks) | 2.25K ops/s | 0.16% | baseline +| `fork_join_lockless/1` | corosio (kqueue) | 46.88K ops/s | 0.48% | -26.6% +| `fork_join_lockless/1` | asio (coroutines) | 29.74K ops/s | 0.48% | -53.4% +| `fork_join_lockless/1` | asio (callbacks) | 63.85K ops/s | 0.50% | baseline +| `fork_join_lockless/16` | corosio (kqueue) | 7.80K ops/s | 0.71% | -4.3% +| `fork_join_lockless/16` | asio (coroutines) | 7.27K ops/s | 0.24% | -10.8% +| `fork_join_lockless/16` | asio (callbacks) | 8.15K ops/s | 0.69% | baseline +| `fork_join_lockless/4` | corosio (kqueue) | 21.77K ops/s | 0.50% | -7.2% +| `fork_join_lockless/4` | asio (coroutines) | 16.89K ops/s | 0.22% | -28.0% +| `fork_join_lockless/4` | asio (callbacks) | 23.47K ops/s | 0.19% | baseline +| `fork_join_lockless/64` | corosio (kqueue) | 2.08K ops/s | 2.38% | -8.5% +| `fork_join_lockless/64` | asio (coroutines) | 2.10K ops/s | 1.02% | -7.4% +| `fork_join_lockless/64` | asio (callbacks) | 2.27K ops/s | 0.13% | baseline +| `nested/16` | corosio (kqueue) | 2.04K ops/s | 1.78% | -8.9% +| `nested/16` | asio (coroutines) | 2.01K ops/s | 0.50% | -10.2% +| `nested/16` | asio (callbacks) | 2.24K ops/s | 0.29% | baseline +| `nested/4` | corosio (kqueue) | 7.07K ops/s | 0.18% | -12.5% +| `nested/4` | asio (coroutines) | 6.49K ops/s | 0.25% | -19.7% +| `nested/4` | asio (callbacks) | 8.08K ops/s | 0.84% | baseline +| `nested_lockless/16` | corosio (kqueue) | 2.00K ops/s | 2.14% | -11.6% +| `nested_lockless/16` | asio (coroutines) | 2.03K ops/s | 0.73% | -10.4% +| `nested_lockless/16` | asio (callbacks) | 2.26K ops/s | 0.25% | baseline +| `nested_lockless/4` | corosio (kqueue) | 7.19K ops/s | 0.41% | -11.8% +| `nested_lockless/4` | asio (coroutines) | 6.56K ops/s | 0.32% | -19.5% +| `nested_lockless/4` | asio (callbacks) | 8.15K ops/s | 0.69% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `http_server` + +Request/response throughput of a minimal HTTP/1.1 server exchanging a fixed small request and canned response over persistent TCP loopback connections. + +[.bch-card] +-- +image::bench/macos-http_server.svg[http server comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (kqueue) | 49.55K ops/s | 0.29% | -17.9% +| `concurrent/1` | asio (coroutines) | 57.16K ops/s | 1.17% | -5.3% +| `concurrent/1` | asio (callbacks) | 60.35K ops/s | 0.94% | baseline +| `concurrent/16` | corosio (kqueue) | 126.9K ops/s | 0.37% | -0.2% +| `concurrent/16` | asio (coroutines) | 126.8K ops/s | 0.63% | -0.3% +| `concurrent/16` | asio (callbacks) | 127.2K ops/s | 0.41% | baseline +| `concurrent/32` | corosio (kqueue) | 127.9K ops/s | 0.92% | -8.4% +| `concurrent/32` | asio (coroutines) | 135.2K ops/s | 0.21% | -3.1% +| `concurrent/32` | asio (callbacks) | 139.6K ops/s | 0.44% | baseline +| `concurrent/4` | corosio (kqueue) | 88.07K ops/s | 0.45% | -4.5% +| `concurrent/4` | asio (coroutines) | 91.01K ops/s | 0.16% | -1.3% +| `concurrent/4` | asio (callbacks) | 92.17K ops/s | 0.35% | baseline +| `multithread/1` | corosio (kqueue) | 127.5K ops/s | 0.90% | -8.6% +| `multithread/1` | asio (coroutines) | 134.9K ops/s | 0.18% | -3.3% +| `multithread/1` | asio (callbacks) | 139.5K ops/s | 0.43% | baseline +| `multithread/16` | corosio (kqueue) | 126.8K ops/s | 0.72% | +9.8% +| `multithread/16` | asio (coroutines) | 114.2K ops/s | 0.49% | -1.1% +| `multithread/16` | asio (callbacks) | 115.5K ops/s | 1.09% | baseline +| `multithread/2` | corosio (kqueue) | 174.1K ops/s | 1.89% | +2.7% +| `multithread/2` | asio (coroutines) | 167.9K ops/s | 0.20% | -1.0% +| `multithread/2` | asio (callbacks) | 169.5K ops/s | 0.13% | baseline +| `multithread/4` | corosio (kqueue) | 162.0K ops/s | 1.02% | +11.7% +| `multithread/4` | asio (coroutines) | 143.3K ops/s | 0.19% | -1.2% +| `multithread/4` | asio (callbacks) | 145.0K ops/s | 0.21% | baseline +| `multithread/8` | corosio (kqueue) | 132.4K ops/s | 1.15% | +0.1% +| `multithread/8` | asio (coroutines) | 128.8K ops/s | 0.57% | -2.6% +| `multithread/8` | asio (callbacks) | 132.2K ops/s | 0.55% | baseline +| `single_conn` | corosio (kqueue) | 49.49K ops/s | 0.20% | -18.2% +| `single_conn` | asio (coroutines) | 56.88K ops/s | 0.89% | -6.0% +| `single_conn` | asio (callbacks) | 60.49K ops/s | 0.80% | baseline +| `single_conn_lockless` | corosio (kqueue) | 49.61K ops/s | 0.34% | -21.0% +| `single_conn_lockless` | asio (coroutines) | 60.26K ops/s | 1.28% | -4.1% +| `single_conn_lockless` | asio (callbacks) | 62.82K ops/s | 0.70% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `local_socket_latency` + +Round-trip latency of a single Unix domain stream socket write/read exchange across message sizes and concurrent pair counts. + +[.bch-card] +-- +image::bench/macos-local_socket_latency.svg[local socket latency comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (kqueue) | 2.06 µs | 0.14% | +95.8% +| `concurrent/1` | asio (coroutines) | 48.60 µs | 0.08% | +0.0% +| `concurrent/1` | asio (callbacks) | 48.62 µs | 0.09% | baseline +| `concurrent/16` | corosio (kqueue) | 30.67 µs | 0.18% | +58.8% +| `concurrent/16` | asio (coroutines) | 77.22 µs | 0.51% | -3.6% +| `concurrent/16` | asio (callbacks) | 74.53 µs | 0.68% | baseline +| `concurrent/4` | corosio (kqueue) | 7.69 µs | 0.28% | +86.1% +| `concurrent/4` | asio (coroutines) | 55.63 µs | 0.21% | -0.5% +| `concurrent/4` | asio (callbacks) | 55.38 µs | 0.19% | baseline +| `concurrent_lockless/1` | corosio (kqueue) | 2.03 µs | 0.17% | +95.8% +| `concurrent_lockless/1` | asio (coroutines) | 48.60 µs | 0.13% | +0.0% +| `concurrent_lockless/1` | asio (callbacks) | 48.61 µs | 0.04% | baseline +| `concurrent_lockless/16` | corosio (kqueue) | 30.25 µs | 0.31% | +58.6% +| `concurrent_lockless/16` | asio (coroutines) | 75.99 µs | 0.92% | -4.1% +| `concurrent_lockless/16` | asio (callbacks) | 73.02 µs | 0.32% | baseline +| `concurrent_lockless/4` | corosio (kqueue) | 7.54 µs | 0.14% | +86.4% +| `concurrent_lockless/4` | asio (coroutines) | 55.41 µs | 0.25% | -0.1% +| `concurrent_lockless/4` | asio (callbacks) | 55.33 µs | 0.49% | baseline +| `pingpong/1` | corosio (kqueue) | 2.06 µs | 0.19% | +95.8% +| `pingpong/1` | asio (coroutines) | 48.62 µs | 0.03% | -0.0% +| `pingpong/1` | asio (callbacks) | 48.61 µs | 0.10% | baseline +| `pingpong/1024` | corosio (kqueue) | 2.16 µs | 0.24% | +95.6% +| `pingpong/1024` | asio (coroutines) | 48.64 µs | 0.15% | -0.1% +| `pingpong/1024` | asio (callbacks) | 48.60 µs | 0.12% | baseline +| `pingpong/64` | corosio (kqueue) | 2.06 µs | 0.36% | +95.8% +| `pingpong/64` | asio (coroutines) | 48.58 µs | 0.03% | +0.0% +| `pingpong/64` | asio (callbacks) | 48.59 µs | 0.14% | baseline +| `pingpong_lockless/1` | corosio (kqueue) | 2.03 µs | 0.15% | +95.8% +| `pingpong_lockless/1` | asio (coroutines) | 48.60 µs | 0.06% | -0.0% +| `pingpong_lockless/1` | asio (callbacks) | 48.58 µs | 0.07% | baseline +| `pingpong_lockless/1024` | corosio (kqueue) | 2.13 µs | 0.28% | +95.6% +| `pingpong_lockless/1024` | asio (coroutines) | 48.60 µs | 0.06% | +0.0% +| `pingpong_lockless/1024` | asio (callbacks) | 48.61 µs | 0.08% | baseline +| `pingpong_lockless/64` | corosio (kqueue) | 2.03 µs | 0.27% | +95.8% +| `pingpong_lockless/64` | asio (coroutines) | 48.58 µs | 0.04% | +0.0% +| `pingpong_lockless/64` | asio (callbacks) | 48.58 µs | 0.04% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `local_socket_throughput` + +Sustained byte throughput of a Unix domain stream socket pair under continuous streaming across chunk sizes and directions. + +[.bch-card] +-- +image::bench/macos-local_socket_throughput-g1.svg[local socket throughput comparison, part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (kqueue) | 1.22 GB/s | 0.26% | +703.2% +| `bidirectional/1024` | asio (coroutines) | 151.7 MB/s | 0.18% | -0.1% +| `bidirectional/1024` | asio (callbacks) | 151.9 MB/s | 0.04% | baseline +| `bidirectional/1048576` | corosio (kqueue) | 3.85 GB/s | 0.45% | +246.9% +| `bidirectional/1048576` | asio (coroutines) | 1.11 GB/s | 0.05% | -0.3% +| `bidirectional/1048576` | asio (callbacks) | 1.11 GB/s | 0.07% | baseline +| `bidirectional/16384` | corosio (kqueue) | 3.85 GB/s | 0.63% | +246.6% +| `bidirectional/16384` | asio (coroutines) | 1.11 GB/s | 0.05% | -0.2% +| `bidirectional/16384` | asio (callbacks) | 1.11 GB/s | 0.38% | baseline +| `bidirectional/262144` | corosio (kqueue) | 3.86 GB/s | 0.72% | +247.3% +| `bidirectional/262144` | asio (coroutines) | 1.11 GB/s | 0.02% | -0.3% +| `bidirectional/262144` | asio (callbacks) | 1.11 GB/s | 0.23% | baseline +| `bidirectional/4096` | corosio (kqueue) | 2.90 GB/s | 0.37% | +377.7% +| `bidirectional/4096` | asio (coroutines) | 605.1 MB/s | 0.30% | -0.3% +| `bidirectional/4096` | asio (callbacks) | 606.7 MB/s | 0.13% | baseline +| `bidirectional/65536` | corosio (kqueue) | 3.85 GB/s | 0.32% | +247.0% +| `bidirectional/65536` | asio (coroutines) | 1.11 GB/s | 0.03% | -0.3% +| `bidirectional/65536` | asio (callbacks) | 1.11 GB/s | 0.31% | baseline +| `bidirectional_lockless/1024` | corosio (kqueue) | 1.23 GB/s | 0.14% | +708.6% +| `bidirectional_lockless/1024` | asio (coroutines) | 151.8 MB/s | 0.05% | -0.1% +| `bidirectional_lockless/1024` | asio (callbacks) | 151.9 MB/s | 0.04% | baseline +| `bidirectional_lockless/1048576` | corosio (kqueue) | 3.92 GB/s | 0.32% | +242.0% +| `bidirectional_lockless/1048576` | asio (coroutines) | 1.11 GB/s | 0.03% | -3.3% +| `bidirectional_lockless/1048576` | asio (callbacks) | 1.14 GB/s | 1.59% | baseline +| `bidirectional_lockless/16384` | corosio (kqueue) | 3.90 GB/s | 0.54% | +240.9% +| `bidirectional_lockless/16384` | asio (coroutines) | 1.11 GB/s | 0.03% | -3.1% +| `bidirectional_lockless/16384` | asio (callbacks) | 1.14 GB/s | 1.35% | baseline +| `bidirectional_lockless/262144` | corosio (kqueue) | 3.90 GB/s | 0.43% | +241.4% +| `bidirectional_lockless/262144` | asio (coroutines) | 1.11 GB/s | 0.02% | -3.1% +| `bidirectional_lockless/262144` | asio (callbacks) | 1.14 GB/s | 1.55% | baseline +| `bidirectional_lockless/4096` | corosio (kqueue) | 2.95 GB/s | 0.38% | +384.9% +| `bidirectional_lockless/4096` | asio (coroutines) | 605.8 MB/s | 0.16% | -0.3% +| `bidirectional_lockless/4096` | asio (callbacks) | 607.4 MB/s | 0.05% | baseline +| `bidirectional_lockless/65536` | corosio (kqueue) | 3.89 GB/s | 0.47% | +242.3% +| `bidirectional_lockless/65536` | asio (coroutines) | 1.11 GB/s | 0.02% | -2.6% +| `bidirectional_lockless/65536` | asio (callbacks) | 1.14 GB/s | 1.77% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/macos-local_socket_throughput-g2.svg[local socket throughput comparison, part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (kqueue) | 1.19 GB/s | 0.36% | +1402.1% +| `unidirectional/1024` | asio (coroutines) | 76.19 MB/s | 0.04% | -4.0% +| `unidirectional/1024` | asio (callbacks) | 79.38 MB/s | 0.75% | baseline +| `unidirectional/1048576` | corosio (kqueue) | 3.46 GB/s | 0.31% | +467.8% +| `unidirectional/1048576` | asio (coroutines) | 608.5 MB/s | 0.08% | -0.1% +| `unidirectional/1048576` | asio (callbacks) | 608.9 MB/s | 0.18% | baseline +| `unidirectional/16384` | corosio (kqueue) | 3.46 GB/s | 0.38% | +467.9% +| `unidirectional/16384` | asio (coroutines) | 608.6 MB/s | 0.03% | -0.0% +| `unidirectional/16384` | asio (callbacks) | 608.7 MB/s | 0.03% | baseline +| `unidirectional/262144` | corosio (kqueue) | 3.45 GB/s | 0.25% | +466.7% +| `unidirectional/262144` | asio (coroutines) | 608.6 MB/s | 0.03% | -0.0% +| `unidirectional/262144` | asio (callbacks) | 608.8 MB/s | 0.03% | baseline +| `unidirectional/4096` | corosio (kqueue) | 2.68 GB/s | 0.21% | +778.7% +| `unidirectional/4096` | asio (coroutines) | 304.5 MB/s | 0.08% | -0.1% +| `unidirectional/4096` | asio (callbacks) | 304.8 MB/s | 0.02% | baseline +| `unidirectional/65536` | corosio (kqueue) | 3.44 GB/s | 0.42% | +465.7% +| `unidirectional/65536` | asio (coroutines) | 608.7 MB/s | 0.02% | -0.0% +| `unidirectional/65536` | asio (callbacks) | 608.8 MB/s | 0.09% | baseline +| `unidirectional_lockless/1024` | corosio (kqueue) | 1.20 GB/s | 0.26% | +1335.5% +| `unidirectional_lockless/1024` | asio (coroutines) | 77.56 MB/s | 0.32% | -7.0% +| `unidirectional_lockless/1024` | asio (callbacks) | 83.40 MB/s | 0.26% | baseline +| `unidirectional_lockless/1048576` | corosio (kqueue) | 3.50 GB/s | 0.43% | +474.9% +| `unidirectional_lockless/1048576` | asio (coroutines) | 608.6 MB/s | 0.04% | -0.1% +| `unidirectional_lockless/1048576` | asio (callbacks) | 609.0 MB/s | 0.03% | baseline +| `unidirectional_lockless/16384` | corosio (kqueue) | 3.50 GB/s | 0.34% | +475.4% +| `unidirectional_lockless/16384` | asio (coroutines) | 608.7 MB/s | 0.01% | -0.0% +| `unidirectional_lockless/16384` | asio (callbacks) | 608.9 MB/s | 0.03% | baseline +| `unidirectional_lockless/262144` | corosio (kqueue) | 3.50 GB/s | 0.34% | +475.1% +| `unidirectional_lockless/262144` | asio (coroutines) | 608.6 MB/s | 0.05% | -0.1% +| `unidirectional_lockless/262144` | asio (callbacks) | 608.9 MB/s | 0.02% | baseline +| `unidirectional_lockless/4096` | corosio (kqueue) | 2.71 GB/s | 0.22% | +776.8% +| `unidirectional_lockless/4096` | asio (coroutines) | 304.6 MB/s | 0.03% | -1.6% +| `unidirectional_lockless/4096` | asio (callbacks) | 309.5 MB/s | 0.58% | baseline +| `unidirectional_lockless/65536` | corosio (kqueue) | 3.50 GB/s | 0.42% | +474.0% +| `unidirectional_lockless/65536` | asio (coroutines) | 608.7 MB/s | 0.02% | -0.0% +| `unidirectional_lockless/65536` | asio (callbacks) | 608.9 MB/s | 0.03% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `socket_latency` + +Round-trip latency of a TCP loopback connection. Each sample is one full round trip (request out, reply back) across message sizes and concurrent pair counts. + +[.bch-card] +-- +image::bench/macos-socket_latency.svg[socket latency comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (kqueue) | 18.82 µs | 0.21% | +64.8% +| `concurrent/1` | asio (coroutines) | 55.96 µs | 2.40% | -4.6% +| `concurrent/1` | asio (callbacks) | 53.52 µs | 4.64% | baseline +| `concurrent/16` | corosio (kqueue) | 120.69 µs | 0.53% | +9.5% +| `concurrent/16` | asio (coroutines) | 142.60 µs | 5.34% | -6.9% +| `concurrent/16` | asio (callbacks) | 133.43 µs | 5.65% | baseline +| `concurrent/4` | corosio (kqueue) | 44.09 µs | 0.77% | +40.3% +| `concurrent/4` | asio (coroutines) | 74.88 µs | 7.32% | -1.3% +| `concurrent/4` | asio (callbacks) | 73.91 µs | 0.80% | baseline +| `concurrent_lockless/1` | corosio (kqueue) | 18.85 µs | 0.49% | +66.7% +| `concurrent_lockless/1` | asio (coroutines) | 55.13 µs | 3.89% | +2.6% +| `concurrent_lockless/1` | asio (callbacks) | 56.61 µs | 6.54% | baseline +| `concurrent_lockless/16` | corosio (kqueue) | 120.26 µs | 0.39% | +8.3% +| `concurrent_lockless/16` | asio (coroutines) | 136.84 µs | 5.71% | -4.3% +| `concurrent_lockless/16` | asio (callbacks) | 131.16 µs | 7.19% | baseline +| `concurrent_lockless/4` | corosio (kqueue) | 43.39 µs | 0.67% | +40.9% +| `concurrent_lockless/4` | asio (coroutines) | 74.37 µs | 13.79% | -1.2% +| `concurrent_lockless/4` | asio (callbacks) | 73.46 µs | 0.93% | baseline +| `pingpong/1` | corosio (kqueue) | 15.56 µs | 0.60% | +68.5% +| `pingpong/1` | asio (coroutines) | 50.73 µs | 1.93% | -2.6% +| `pingpong/1` | asio (callbacks) | 49.42 µs | 2.41% | baseline +| `pingpong/1024` | corosio (kqueue) | 19.02 µs | 0.25% | +66.2% +| `pingpong/1024` | asio (coroutines) | 53.34 µs | 6.26% | +5.1% +| `pingpong/1024` | asio (callbacks) | 56.18 µs | 5.76% | baseline +| `pingpong/64` | corosio (kqueue) | 18.81 µs | 0.51% | +66.8% +| `pingpong/64` | asio (coroutines) | 56.64 µs | 3.02% | -0.0% +| `pingpong/64` | asio (callbacks) | 56.62 µs | 4.45% | baseline +| `pingpong_lockless/1` | corosio (kqueue) | 15.47 µs | 0.43% | +69.7% +| `pingpong_lockless/1` | asio (coroutines) | 55.05 µs | 5.28% | -7.8% +| `pingpong_lockless/1` | asio (callbacks) | 51.09 µs | 5.04% | baseline +| `pingpong_lockless/1024` | corosio (kqueue) | 19.03 µs | 0.36% | +66.4% +| `pingpong_lockless/1024` | asio (coroutines) | 56.89 µs | 2.43% | -0.4% +| `pingpong_lockless/1024` | asio (callbacks) | 56.67 µs | 2.05% | baseline +| `pingpong_lockless/64` | corosio (kqueue) | 18.85 µs | 0.41% | +66.8% +| `pingpong_lockless/64` | asio (coroutines) | 56.61 µs | 3.37% | +0.4% +| `pingpong_lockless/64` | asio (callbacks) | 56.84 µs | 4.40% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `socket_throughput` + +Sustained byte throughput of a TCP loopback connection under continuous streaming, varying chunk size, direction, and concurrency. + +[.bch-card] +-- +image::bench/macos-socket_throughput-g1.svg[socket throughput comparison, part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (kqueue) | 410.3 MB/s | 0.65% | +100.0% +| `bidirectional/1024` | asio (coroutines) | 198.8 MB/s | 0.87% | -3.1% +| `bidirectional/1024` | asio (callbacks) | 205.1 MB/s | 0.67% | baseline +| `bidirectional/1048576` | corosio (kqueue) | 7.75 GB/s | 0.71% | -11.6% +| `bidirectional/1048576` | asio (coroutines) | 8.57 GB/s | 1.62% | -2.2% +| `bidirectional/1048576` | asio (callbacks) | 8.77 GB/s | 2.18% | baseline +| `bidirectional/16384` | corosio (kqueue) | 4.01 GB/s | 0.58% | +64.9% +| `bidirectional/16384` | asio (coroutines) | 2.39 GB/s | 0.37% | -1.8% +| `bidirectional/16384` | asio (callbacks) | 2.43 GB/s | 0.31% | baseline +| `bidirectional/262144` | corosio (kqueue) | 9.02 GB/s | 1.59% | -1.9% +| `bidirectional/262144` | asio (coroutines) | 7.37 GB/s | 8.49% | -19.9% +| `bidirectional/262144` | asio (callbacks) | 9.20 GB/s | 0.56% | baseline +| `bidirectional/4096` | corosio (kqueue) | 1.49 GB/s | 0.30% | +88.6% +| `bidirectional/4096` | asio (coroutines) | 756.1 MB/s | 0.87% | -4.2% +| `bidirectional/4096` | asio (callbacks) | 789.4 MB/s | 0.68% | baseline +| `bidirectional/65536` | corosio (kqueue) | 7.59 GB/s | 1.46% | +32.1% +| `bidirectional/65536` | asio (coroutines) | 5.74 GB/s | 1.55% | -0.2% +| `bidirectional/65536` | asio (callbacks) | 5.75 GB/s | 0.86% | baseline +| `bidirectional_lockless/1024` | corosio (kqueue) | 411.7 MB/s | 0.57% | +104.4% +| `bidirectional_lockless/1024` | asio (coroutines) | 201.3 MB/s | 0.67% | -0.1% +| `bidirectional_lockless/1024` | asio (callbacks) | 201.4 MB/s | 1.34% | baseline +| `bidirectional_lockless/1048576` | corosio (kqueue) | 7.76 GB/s | 2.00% | -11.3% +| `bidirectional_lockless/1048576` | asio (coroutines) | 8.72 GB/s | 2.31% | -0.3% +| `bidirectional_lockless/1048576` | asio (callbacks) | 8.75 GB/s | 2.38% | baseline +| `bidirectional_lockless/16384` | corosio (kqueue) | 4.00 GB/s | 0.47% | +63.3% +| `bidirectional_lockless/16384` | asio (coroutines) | 2.40 GB/s | 0.51% | -1.9% +| `bidirectional_lockless/16384` | asio (callbacks) | 2.45 GB/s | 0.22% | baseline +| `bidirectional_lockless/262144` | corosio (kqueue) | 8.95 GB/s | 1.93% | -3.8% +| `bidirectional_lockless/262144` | asio (coroutines) | 7.43 GB/s | 10.71% | -20.2% +| `bidirectional_lockless/262144` | asio (callbacks) | 9.31 GB/s | 7.51% | baseline +| `bidirectional_lockless/4096` | corosio (kqueue) | 1.50 GB/s | 0.51% | +86.4% +| `bidirectional_lockless/4096` | asio (coroutines) | 773.4 MB/s | 0.50% | -3.6% +| `bidirectional_lockless/4096` | asio (callbacks) | 802.7 MB/s | 1.08% | baseline +| `bidirectional_lockless/65536` | corosio (kqueue) | 7.82 GB/s | 1.25% | +36.1% +| `bidirectional_lockless/65536` | asio (coroutines) | 5.74 GB/s | 1.34% | -0.0% +| `bidirectional_lockless/65536` | asio (callbacks) | 5.75 GB/s | 1.66% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/macos-socket_throughput-g2.svg[socket throughput comparison, part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `multithread/2` | corosio (kqueue) | 7.75 GB/s | 1.43% | +2.4% +| `multithread/2` | asio (coroutines) | 7.76 GB/s | 1.90% | +2.4% +| `multithread/2` | asio (callbacks) | 7.57 GB/s | 2.22% | baseline +| `multithread/4` | corosio (kqueue) | 3.92 GB/s | 2.78% | +0.6% +| `multithread/4` | asio (coroutines) | 4.08 GB/s | 1.87% | +4.8% +| `multithread/4` | asio (callbacks) | 3.90 GB/s | 22.01% | baseline +| `multithread/8` | corosio (kqueue) | 3.48 GB/s | 3.73% | -10.8% +| `multithread/8` | asio (coroutines) | 3.91 GB/s | 3.00% | +0.2% +| `multithread/8` | asio (callbacks) | 3.90 GB/s | 2.02% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/macos-socket_throughput-g3.svg[socket throughput comparison, part 3,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (kqueue) | 321.5 MB/s | 1.01% | +94.3% +| `unidirectional/1024` | asio (coroutines) | 157.8 MB/s | 0.52% | -4.7% +| `unidirectional/1024` | asio (callbacks) | 165.5 MB/s | 2.47% | baseline +| `unidirectional/1048576` | corosio (kqueue) | 9.05 GB/s | 1.04% | -0.1% +| `unidirectional/1048576` | asio (coroutines) | 8.34 GB/s | 5.41% | -8.0% +| `unidirectional/1048576` | asio (callbacks) | 9.07 GB/s | 1.31% | baseline +| `unidirectional/16384` | corosio (kqueue) | 3.75 GB/s | 1.25% | +96.0% +| `unidirectional/16384` | asio (coroutines) | 1.90 GB/s | 0.43% | -0.8% +| `unidirectional/16384` | asio (callbacks) | 1.92 GB/s | 1.18% | baseline +| `unidirectional/262144` | corosio (kqueue) | 8.96 GB/s | 2.20% | +14.0% +| `unidirectional/262144` | asio (coroutines) | 6.07 GB/s | 8.74% | -22.7% +| `unidirectional/262144` | asio (callbacks) | 7.86 GB/s | 2.31% | baseline +| `unidirectional/4096` | corosio (kqueue) | 1.24 GB/s | 1.29% | +97.7% +| `unidirectional/4096` | asio (coroutines) | 602.1 MB/s | 0.54% | -4.4% +| `unidirectional/4096` | asio (callbacks) | 629.7 MB/s | 1.12% | baseline +| `unidirectional/65536` | corosio (kqueue) | 7.32 GB/s | 1.44% | +67.2% +| `unidirectional/65536` | asio (coroutines) | 4.26 GB/s | 1.87% | -2.6% +| `unidirectional/65536` | asio (callbacks) | 4.37 GB/s | 2.43% | baseline +| `unidirectional_lockless/1024` | corosio (kqueue) | 328.2 MB/s | 0.51% | +125.2% +| `unidirectional_lockless/1024` | asio (coroutines) | 159.8 MB/s | 0.58% | +9.6% +| `unidirectional_lockless/1024` | asio (callbacks) | 145.7 MB/s | 0.59% | baseline +| `unidirectional_lockless/1048576` | corosio (kqueue) | 9.12 GB/s | 0.81% | +1.2% +| `unidirectional_lockless/1048576` | asio (coroutines) | 8.25 GB/s | 5.89% | -8.5% +| `unidirectional_lockless/1048576` | asio (callbacks) | 9.02 GB/s | 1.13% | baseline +| `unidirectional_lockless/16384` | corosio (kqueue) | 3.76 GB/s | 0.54% | +94.8% +| `unidirectional_lockless/16384` | asio (coroutines) | 1.90 GB/s | 0.94% | -1.8% +| `unidirectional_lockless/16384` | asio (callbacks) | 1.93 GB/s | 1.53% | baseline +| `unidirectional_lockless/262144` | corosio (kqueue) | 9.14 GB/s | 1.74% | +20.4% +| `unidirectional_lockless/262144` | asio (coroutines) | 6.07 GB/s | 8.88% | -20.0% +| `unidirectional_lockless/262144` | asio (callbacks) | 7.59 GB/s | 10.28% | baseline +| `unidirectional_lockless/4096` | corosio (kqueue) | 1.26 GB/s | 0.74% | +96.5% +| `unidirectional_lockless/4096` | asio (coroutines) | 613.2 MB/s | 0.64% | -4.1% +| `unidirectional_lockless/4096` | asio (callbacks) | 639.1 MB/s | 0.40% | baseline +| `unidirectional_lockless/65536` | corosio (kqueue) | 7.39 GB/s | 0.88% | +74.6% +| `unidirectional_lockless/65536` | asio (coroutines) | 4.20 GB/s | 1.66% | -0.7% +| `unidirectional_lockless/65536` | asio (callbacks) | 4.23 GB/s | 2.96% | baseline +|=== +==== + +-- + diff --git a/doc/modules/ROOT/pages/benchmarks/windows.adoc b/doc/modules/ROOT/pages/benchmarks/windows.adoc new file mode 100644 index 000000000..2fd055bf0 --- /dev/null +++ b/doc/modules/ROOT/pages/benchmarks/windows.adoc @@ -0,0 +1,426 @@ += Windows Benchmarks +:page-mode: explanation +:toc: left + +_Generated 2026-10-02 from corosio `e2de06fda069` — https://github.com/cppalliance/corosio/actions/runs/37015995763[benchmark run]. This page is fully generated; do not edit by hand (see the xref:benchmarks/index.adoc[Benchmarks landing page])._ + +== Summary + +_Within noise_ means the relative difference is within twice the combined run-to-run noise (the root-sum-square of each side's CV). Differences that small are indistinguishable from measurement jitter. See xref:benchmarks/index.adoc[Methodology] for how these figures are computed. + +**0 faster · 18 within noise · 59 slower** of 77 benchmarks — median **-6.1%** vs Boost.Asio (callbacks) on the same reactor. + +image::bench/windows-summary.svg[summary,role=bch-block,opts=inline] + +== Test Environment + +[cols="1,3"] +|=== +| CPU | AMD Ryzen 5 3600 6-Core Processor +| Cores | 6 +| RAM (GB) | 64 +| OS | MSYS_NT-10.0-20348 3.4.7-ea781829.x86_64 +| Kernel/build | 2023-07-05 12:05 UTC +| Compiler | c++.exe (x86_64-posix-seh-rev2, Built by MinGW-W64 project) 12.2.0 +| CMake | `cmake version 3.26.3` +| liburing | n/a +| Boost commit | 39fa1e9a3491bd099b570935b3f3422065f91b03 +| Asio commit | a7dc25b4cb6c49a6946d86ea20664f1027203225 +| Asio reactor | IOCP +| Capy commit | a372a6b054261f29497ac0d19c2a9533f9eaad40 +| Corosio commit | e2de06fda069607f34bc957cbefcbd9ba12fddca +| Corosio branch | pr/benchmark-report +| Date (UTC) | 2026-10-02 +| Iterations | 7 +| Duration per benchmark (s) | 2.0 +|=== + +== Results + +=== `accept_churn` + +Rate of setting up and tearing down short-lived TCP connections: connect, accept, close. + +[.bch-card] +-- +image::bench/windows-accept_churn.svg[accept churn comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `burst/10` | corosio (IOCP) | 1.54K ops/s | 0.38% | -7.8% +| `burst/10` | asio (coroutines) | 1.70K ops/s | 0.57% | +2.2% +| `burst/10` | asio (callbacks) | 1.67K ops/s | 0.65% | baseline +| `burst/100` | corosio (IOCP) | 149.4 ops/s | 0.43% | -8.7% +| `burst/100` | asio (coroutines) | 171.8 ops/s | 0.54% | +4.9% +| `burst/100` | asio (callbacks) | 163.8 ops/s | 0.68% | baseline +| `burst_lockless/10` | corosio (IOCP) | 1.54K ops/s | 0.26% | -7.8% +| `burst_lockless/10` | asio (coroutines) | 1.72K ops/s | 0.23% | +2.5% +| `burst_lockless/10` | asio (callbacks) | 1.67K ops/s | 0.23% | baseline +| `burst_lockless/100` | corosio (IOCP) | 154.4 ops/s | 0.73% | -7.7% +| `burst_lockless/100` | asio (coroutines) | 177.1 ops/s | 0.53% | +5.9% +| `burst_lockless/100` | asio (callbacks) | 167.2 ops/s | 0.48% | baseline +| `concurrent/1` | corosio (IOCP) | 12.38K ops/s | 0.31% | -9.8% +| `concurrent/1` | asio (coroutines) | 13.22K ops/s | 0.09% | -3.6% +| `concurrent/1` | asio (callbacks) | 13.72K ops/s | 0.18% | baseline +| `concurrent/16` | corosio (IOCP) | 13.30K ops/s | 0.24% | -7.9% +| `concurrent/16` | asio (coroutines) | 14.00K ops/s | 0.22% | -3.0% +| `concurrent/16` | asio (callbacks) | 14.44K ops/s | 0.17% | baseline +| `concurrent/4` | corosio (IOCP) | 13.02K ops/s | 0.26% | -8.5% +| `concurrent/4` | asio (coroutines) | 13.81K ops/s | 0.19% | -3.0% +| `concurrent/4` | asio (callbacks) | 14.24K ops/s | 0.33% | baseline +| `sequential` | corosio (IOCP) | 12.34K ops/s | 1.24% | -9.7% +| `sequential` | asio (coroutines) | 13.15K ops/s | 0.17% | -3.8% +| `sequential` | asio (callbacks) | 13.67K ops/s | 0.18% | baseline +| `sequential_lockless` | corosio (IOCP) | 12.39K ops/s | 1.32% | -9.5% +| `sequential_lockless` | asio (coroutines) | 13.20K ops/s | 0.16% | -3.7% +| `sequential_lockless` | asio (callbacks) | 13.70K ops/s | 0.15% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `fan_out` + +Fan-out/fan-in coroutine coordination: a parent starts concurrent sub-requests against echo servers and awaits their completion via a shared latch. + +[.bch-card] +-- +image::bench/windows-fan_out-g1.svg[fan out comparison, part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent_parents/1` | corosio (IOCP) | 7.16K ops/s | 0.30% | -7.4% +| `concurrent_parents/1` | asio (coroutines) | 6.85K ops/s | 0.38% | -11.3% +| `concurrent_parents/1` | asio (callbacks) | 7.73K ops/s | 0.21% | baseline +| `concurrent_parents/16` | corosio (IOCP) | 6.43K ops/s | 0.76% | -10.7% +| `concurrent_parents/16` | asio (coroutines) | 6.37K ops/s | 0.73% | -11.5% +| `concurrent_parents/16` | asio (callbacks) | 7.20K ops/s | 0.43% | baseline +| `concurrent_parents/4` | corosio (IOCP) | 6.90K ops/s | 0.36% | -9.1% +| `concurrent_parents/4` | asio (coroutines) | 6.81K ops/s | 0.34% | -10.2% +| `concurrent_parents/4` | asio (callbacks) | 7.59K ops/s | 0.29% | baseline +| `concurrent_parents_lockless/1` | corosio (IOCP) | 7.15K ops/s | 0.22% | -7.5% +| `concurrent_parents_lockless/1` | asio (coroutines) | 6.95K ops/s | 0.23% | -10.1% +| `concurrent_parents_lockless/1` | asio (callbacks) | 7.73K ops/s | 0.16% | baseline +| `concurrent_parents_lockless/16` | corosio (IOCP) | 6.44K ops/s | 0.63% | -11.1% +| `concurrent_parents_lockless/16` | asio (coroutines) | 6.42K ops/s | 0.34% | -11.3% +| `concurrent_parents_lockless/16` | asio (callbacks) | 7.24K ops/s | 0.65% | baseline +| `concurrent_parents_lockless/4` | corosio (IOCP) | 6.92K ops/s | 0.34% | -9.1% +| `concurrent_parents_lockless/4` | asio (coroutines) | 6.80K ops/s | 0.23% | -10.5% +| `concurrent_parents_lockless/4` | asio (callbacks) | 7.61K ops/s | 0.16% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/windows-fan_out-g2.svg[fan out comparison, part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `fork_join/1` | corosio (IOCP) | 112.4K ops/s | 0.37% | -8.9% +| `fork_join/1` | asio (coroutines) | 109.9K ops/s | 0.96% | -11.0% +| `fork_join/1` | asio (callbacks) | 123.5K ops/s | 0.27% | baseline +| `fork_join/16` | corosio (IOCP) | 7.15K ops/s | 0.14% | -7.6% +| `fork_join/16` | asio (coroutines) | 6.85K ops/s | 0.87% | -11.5% +| `fork_join/16` | asio (callbacks) | 7.74K ops/s | 0.18% | baseline +| `fork_join/4` | corosio (IOCP) | 29.33K ops/s | 0.26% | -4.8% +| `fork_join/4` | asio (coroutines) | 27.75K ops/s | 0.28% | -9.9% +| `fork_join/4` | asio (callbacks) | 30.82K ops/s | 0.17% | baseline +| `fork_join/64` | corosio (IOCP) | 1.73K ops/s | 0.83% | -8.7% +| `fork_join/64` | asio (coroutines) | 1.70K ops/s | 0.17% | -9.9% +| `fork_join/64` | asio (callbacks) | 1.89K ops/s | 0.46% | baseline +| `fork_join_lockless/1` | corosio (IOCP) | 112.9K ops/s | 0.19% | -8.8% +| `fork_join_lockless/1` | asio (coroutines) | 109.8K ops/s | 0.53% | -11.3% +| `fork_join_lockless/1` | asio (callbacks) | 123.7K ops/s | 0.39% | baseline +| `fork_join_lockless/16` | corosio (IOCP) | 7.16K ops/s | 0.14% | -7.4% +| `fork_join_lockless/16` | asio (coroutines) | 6.86K ops/s | 0.51% | -11.3% +| `fork_join_lockless/16` | asio (callbacks) | 7.73K ops/s | 0.26% | baseline +| `fork_join_lockless/4` | corosio (IOCP) | 29.24K ops/s | 0.33% | -5.1% +| `fork_join_lockless/4` | asio (coroutines) | 27.80K ops/s | 0.19% | -9.8% +| `fork_join_lockless/4` | asio (callbacks) | 30.81K ops/s | 0.16% | baseline +| `fork_join_lockless/64` | corosio (IOCP) | 1.73K ops/s | 0.52% | -8.8% +| `fork_join_lockless/64` | asio (coroutines) | 1.71K ops/s | 0.37% | -9.8% +| `fork_join_lockless/64` | asio (callbacks) | 1.90K ops/s | 0.42% | baseline +| `nested/16` | corosio (IOCP) | 1.70K ops/s | 0.56% | -10.5% +| `nested/16` | asio (coroutines) | 1.66K ops/s | 0.39% | -12.3% +| `nested/16` | asio (callbacks) | 1.89K ops/s | 0.26% | baseline +| `nested/4` | corosio (IOCP) | 6.99K ops/s | 0.24% | -8.9% +| `nested/4` | asio (coroutines) | 6.65K ops/s | 0.53% | -13.3% +| `nested/4` | asio (callbacks) | 7.67K ops/s | 0.52% | baseline +| `nested_lockless/16` | corosio (IOCP) | 1.69K ops/s | 0.40% | -10.7% +| `nested_lockless/16` | asio (coroutines) | 1.66K ops/s | 0.49% | -12.7% +| `nested_lockless/16` | asio (callbacks) | 1.90K ops/s | 0.34% | baseline +| `nested_lockless/4` | corosio (IOCP) | 7.00K ops/s | 0.30% | -9.0% +| `nested_lockless/4` | asio (coroutines) | 6.66K ops/s | 0.42% | -13.4% +| `nested_lockless/4` | asio (callbacks) | 7.69K ops/s | 0.63% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `http_server` + +Request/response throughput of a minimal HTTP/1.1 server exchanging a fixed small request and canned response over persistent TCP loopback connections. + +[.bch-card] +-- +image::bench/windows-http_server.svg[http server comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (IOCP) | 111.9K ops/s | 0.42% | -0.8% +| `concurrent/1` | asio (coroutines) | 107.8K ops/s | 0.33% | -4.5% +| `concurrent/1` | asio (callbacks) | 112.8K ops/s | 0.37% | baseline +| `concurrent/16` | corosio (IOCP) | 108.6K ops/s | 0.33% | -1.9% +| `concurrent/16` | asio (coroutines) | 106.2K ops/s | 0.39% | -4.0% +| `concurrent/16` | asio (callbacks) | 110.7K ops/s | 0.23% | baseline +| `concurrent/32` | corosio (IOCP) | 108.1K ops/s | 0.20% | -1.7% +| `concurrent/32` | asio (coroutines) | 105.6K ops/s | 0.15% | -4.0% +| `concurrent/32` | asio (callbacks) | 109.9K ops/s | 0.24% | baseline +| `concurrent/4` | corosio (IOCP) | 109.6K ops/s | 0.24% | -1.1% +| `concurrent/4` | asio (coroutines) | 106.4K ops/s | 0.26% | -4.0% +| `concurrent/4` | asio (callbacks) | 110.8K ops/s | 0.35% | baseline +| `multithread/1` | corosio (IOCP) | 108.0K ops/s | 0.17% | -1.6% +| `multithread/1` | asio (coroutines) | 105.6K ops/s | 0.24% | -3.7% +| `multithread/1` | asio (callbacks) | 109.7K ops/s | 0.26% | baseline +| `multithread/16` | corosio (IOCP) | 268.9K ops/s | 0.21% | -6.1% +| `multithread/16` | asio (coroutines) | 279.3K ops/s | 0.43% | -2.4% +| `multithread/16` | asio (callbacks) | 286.3K ops/s | 0.52% | baseline +| `multithread/2` | corosio (IOCP) | 181.9K ops/s | 0.68% | -2.9% +| `multithread/2` | asio (coroutines) | 180.1K ops/s | 0.51% | -3.9% +| `multithread/2` | asio (callbacks) | 187.4K ops/s | 0.47% | baseline +| `multithread/4` | corosio (IOCP) | 225.9K ops/s | 0.37% | -4.5% +| `multithread/4` | asio (coroutines) | 228.7K ops/s | 0.65% | -3.3% +| `multithread/4` | asio (callbacks) | 236.6K ops/s | 0.55% | baseline +| `multithread/8` | corosio (IOCP) | 268.0K ops/s | 0.42% | -5.8% +| `multithread/8` | asio (coroutines) | 277.0K ops/s | 0.56% | -2.6% +| `multithread/8` | asio (callbacks) | 284.6K ops/s | 0.68% | baseline +| `single_conn` | corosio (IOCP) | 111.8K ops/s | 0.37% | -0.7% +| `single_conn` | asio (coroutines) | 107.8K ops/s | 0.05% | -4.2% +| `single_conn` | asio (callbacks) | 112.5K ops/s | 0.26% | baseline +| `single_conn_lockless` | corosio (IOCP) | 112.2K ops/s | 0.32% | -0.5% +| `single_conn_lockless` | asio (coroutines) | 107.6K ops/s | 0.56% | -4.5% +| `single_conn_lockless` | asio (callbacks) | 112.7K ops/s | 0.20% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `socket_latency` + +Round-trip latency of a TCP loopback connection. Each sample is one full round trip (request out, reply back) across message sizes and concurrent pair counts. + +[.bch-card] +-- +image::bench/windows-socket_latency.svg[socket latency comparison,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `concurrent/1` | corosio (IOCP) | 7.57 µs | 0.40% | -7.2% +| `concurrent/1` | asio (coroutines) | 7.48 µs | 0.45% | -5.9% +| `concurrent/1` | asio (callbacks) | 7.06 µs | 0.17% | baseline +| `concurrent/16` | corosio (IOCP) | 130.96 µs | 6.14% | -11.2% +| `concurrent/16` | asio (coroutines) | 127.44 µs | 2.57% | -8.2% +| `concurrent/16` | asio (callbacks) | 117.75 µs | 3.93% | baseline +| `concurrent/4` | corosio (IOCP) | 30.40 µs | 0.21% | -7.4% +| `concurrent/4` | asio (coroutines) | 29.95 µs | 0.34% | -5.9% +| `concurrent/4` | asio (callbacks) | 28.30 µs | 0.28% | baseline +| `concurrent_lockless/1` | corosio (IOCP) | 7.57 µs | 0.56% | -7.3% +| `concurrent_lockless/1` | asio (coroutines) | 7.48 µs | 0.22% | -6.0% +| `concurrent_lockless/1` | asio (callbacks) | 7.06 µs | 0.17% | baseline +| `concurrent_lockless/16` | corosio (IOCP) | 121.42 µs | 0.27% | -8.0% +| `concurrent_lockless/16` | asio (coroutines) | 120.27 µs | 0.37% | -7.0% +| `concurrent_lockless/16` | asio (callbacks) | 112.43 µs | 0.28% | baseline +| `concurrent_lockless/4` | corosio (IOCP) | 30.43 µs | 0.42% | -7.6% +| `concurrent_lockless/4` | asio (coroutines) | 29.95 µs | 0.41% | -5.9% +| `concurrent_lockless/4` | asio (callbacks) | 28.28 µs | 0.37% | baseline +| `pingpong/1` | corosio (IOCP) | 7.56 µs | 0.44% | -7.0% +| `pingpong/1` | asio (coroutines) | 7.45 µs | 0.77% | -5.5% +| `pingpong/1` | asio (callbacks) | 7.06 µs | 0.28% | baseline +| `pingpong/1024` | corosio (IOCP) | 7.67 µs | 0.48% | -7.2% +| `pingpong/1024` | asio (coroutines) | 7.56 µs | 0.37% | -5.7% +| `pingpong/1024` | asio (callbacks) | 7.16 µs | 0.49% | baseline +| `pingpong/64` | corosio (IOCP) | 7.60 µs | 0.41% | -7.5% +| `pingpong/64` | asio (coroutines) | 7.46 µs | 0.22% | -5.5% +| `pingpong/64` | asio (callbacks) | 7.07 µs | 0.22% | baseline +| `pingpong_lockless/1` | corosio (IOCP) | 7.53 µs | 0.25% | -7.0% +| `pingpong_lockless/1` | asio (coroutines) | 7.43 µs | 0.27% | -5.6% +| `pingpong_lockless/1` | asio (callbacks) | 7.04 µs | 0.35% | baseline +| `pingpong_lockless/1024` | corosio (IOCP) | 7.69 µs | 0.25% | -7.5% +| `pingpong_lockless/1024` | asio (coroutines) | 7.56 µs | 0.36% | -5.6% +| `pingpong_lockless/1024` | asio (callbacks) | 7.15 µs | 0.35% | baseline +| `pingpong_lockless/64` | corosio (IOCP) | 7.56 µs | 0.17% | -7.0% +| `pingpong_lockless/64` | asio (coroutines) | 7.47 µs | 0.29% | -5.7% +| `pingpong_lockless/64` | asio (callbacks) | 7.07 µs | 0.18% | baseline +|=== +==== + +-- + +++++ +
+++++ + +=== `socket_throughput` + +Sustained byte throughput of a TCP loopback connection under continuous streaming, varying chunk size, direction, and concurrency. + +[.bch-card] +-- +image::bench/windows-socket_throughput-g1.svg[socket throughput comparison, part 1,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `bidirectional/1024` | corosio (IOCP) | 279.9 MB/s | 0.32% | -3.0% +| `bidirectional/1024` | asio (coroutines) | 274.1 MB/s | 0.34% | -5.0% +| `bidirectional/1024` | asio (callbacks) | 288.6 MB/s | 0.51% | baseline +| `bidirectional/1048576` | corosio (IOCP) | 4.42 GB/s | 7.80% | -9.0% +| `bidirectional/1048576` | asio (coroutines) | 4.66 GB/s | 6.82% | -4.2% +| `bidirectional/1048576` | asio (callbacks) | 4.86 GB/s | 9.63% | baseline +| `bidirectional/16384` | corosio (IOCP) | 3.75 GB/s | 0.61% | -1.9% +| `bidirectional/16384` | asio (coroutines) | 3.68 GB/s | 0.34% | -3.8% +| `bidirectional/16384` | asio (callbacks) | 3.82 GB/s | 0.55% | baseline +| `bidirectional/262144` | corosio (IOCP) | 9.55 GB/s | 0.83% | +0.2% +| `bidirectional/262144` | asio (coroutines) | 9.47 GB/s | 6.89% | -0.7% +| `bidirectional/262144` | asio (callbacks) | 9.54 GB/s | 0.42% | baseline +| `bidirectional/4096` | corosio (IOCP) | 1.07 GB/s | 0.27% | -2.8% +| `bidirectional/4096` | asio (coroutines) | 1.06 GB/s | 0.45% | -4.3% +| `bidirectional/4096` | asio (callbacks) | 1.10 GB/s | 0.36% | baseline +| `bidirectional/65536` | corosio (IOCP) | 9.04 GB/s | 0.94% | -2.6% +| `bidirectional/65536` | asio (coroutines) | 9.00 GB/s | 2.37% | -3.0% +| `bidirectional/65536` | asio (callbacks) | 9.28 GB/s | 0.96% | baseline +| `bidirectional_lockless/1024` | corosio (IOCP) | 279.6 MB/s | 0.44% | -2.9% +| `bidirectional_lockless/1024` | asio (coroutines) | 274.6 MB/s | 0.65% | -4.6% +| `bidirectional_lockless/1024` | asio (callbacks) | 287.9 MB/s | 0.49% | baseline +| `bidirectional_lockless/1048576` | corosio (IOCP) | 4.69 GB/s | 7.57% | -1.5% +| `bidirectional_lockless/1048576` | asio (coroutines) | 4.61 GB/s | 7.67% | -3.1% +| `bidirectional_lockless/1048576` | asio (callbacks) | 4.76 GB/s | 8.69% | baseline +| `bidirectional_lockless/16384` | corosio (IOCP) | 3.72 GB/s | 0.85% | -1.8% +| `bidirectional_lockless/16384` | asio (coroutines) | 3.68 GB/s | 0.61% | -3.0% +| `bidirectional_lockless/16384` | asio (callbacks) | 3.79 GB/s | 0.23% | baseline +| `bidirectional_lockless/262144` | corosio (IOCP) | 9.76 GB/s | 1.34% | -0.1% +| `bidirectional_lockless/262144` | asio (coroutines) | 9.71 GB/s | 0.99% | -0.6% +| `bidirectional_lockless/262144` | asio (callbacks) | 9.76 GB/s | 1.25% | baseline +| `bidirectional_lockless/4096` | corosio (IOCP) | 1.07 GB/s | 0.29% | -3.0% +| `bidirectional_lockless/4096` | asio (coroutines) | 1.06 GB/s | 0.52% | -4.6% +| `bidirectional_lockless/4096` | asio (callbacks) | 1.11 GB/s | 0.47% | baseline +| `bidirectional_lockless/65536` | corosio (IOCP) | 8.91 GB/s | 1.43% | -2.0% +| `bidirectional_lockless/65536` | asio (coroutines) | 8.85 GB/s | 0.38% | -2.7% +| `bidirectional_lockless/65536` | asio (callbacks) | 9.09 GB/s | 0.19% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/windows-socket_throughput-g2.svg[socket throughput comparison, part 2,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `multithread/2` | corosio (IOCP) | 12.68 GB/s | 0.54% | -2.1% +| `multithread/2` | asio (coroutines) | 12.73 GB/s | 0.28% | -1.8% +| `multithread/2` | asio (callbacks) | 12.96 GB/s | 0.37% | baseline +| `multithread/4` | corosio (IOCP) | 17.09 GB/s | 0.54% | -2.7% +| `multithread/4` | asio (coroutines) | 17.06 GB/s | 0.77% | -2.8% +| `multithread/4` | asio (callbacks) | 17.56 GB/s | 0.45% | baseline +| `multithread/8` | corosio (IOCP) | 18.40 GB/s | 0.37% | -0.4% +| `multithread/8` | asio (coroutines) | 18.38 GB/s | 0.98% | -0.6% +| `multithread/8` | asio (callbacks) | 18.48 GB/s | 1.82% | baseline +|=== +==== + +-- + +[.bch-card] +-- +image::bench/windows-socket_throughput-g3.svg[socket throughput comparison, part 3,role=bch-block,opts=inline] + +.Detailed results +[%collapsible.bch-exact] +==== +[%autowidth.stretch,options="header"] +|=== +| Benchmark | Implementation | Median | CV | vs asio +| `unidirectional/1024` | corosio (IOCP) | 283.5 MB/s | 0.40% | -2.7% +| `unidirectional/1024` | asio (coroutines) | 279.4 MB/s | 0.56% | -4.1% +| `unidirectional/1024` | asio (callbacks) | 291.4 MB/s | 0.32% | baseline +| `unidirectional/1048576` | corosio (IOCP) | 3.93 GB/s | 3.90% | +0.5% +| `unidirectional/1048576` | asio (coroutines) | 3.94 GB/s | 2.24% | +0.6% +| `unidirectional/1048576` | asio (callbacks) | 3.91 GB/s | 5.15% | baseline +| `unidirectional/16384` | corosio (IOCP) | 3.78 GB/s | 1.35% | -2.3% +| `unidirectional/16384` | asio (coroutines) | 3.73 GB/s | 0.73% | -3.5% +| `unidirectional/16384` | asio (callbacks) | 3.87 GB/s | 0.66% | baseline +| `unidirectional/262144` | corosio (IOCP) | 8.75 GB/s | 7.22% | -1.2% +| `unidirectional/262144` | asio (coroutines) | 8.54 GB/s | 4.50% | -3.6% +| `unidirectional/262144` | asio (callbacks) | 8.86 GB/s | 3.82% | baseline +| `unidirectional/4096` | corosio (IOCP) | 1.09 GB/s | 0.43% | -2.5% +| `unidirectional/4096` | asio (coroutines) | 1.07 GB/s | 0.12% | -4.0% +| `unidirectional/4096` | asio (callbacks) | 1.12 GB/s | 0.44% | baseline +| `unidirectional/65536` | corosio (IOCP) | 9.27 GB/s | 3.59% | -1.9% +| `unidirectional/65536` | asio (coroutines) | 9.23 GB/s | 1.69% | -2.4% +| `unidirectional/65536` | asio (callbacks) | 9.45 GB/s | 1.78% | baseline +| `unidirectional_lockless/1024` | corosio (IOCP) | 282.6 MB/s | 0.33% | -2.6% +| `unidirectional_lockless/1024` | asio (coroutines) | 279.3 MB/s | 0.50% | -3.8% +| `unidirectional_lockless/1024` | asio (callbacks) | 290.2 MB/s | 0.42% | baseline +| `unidirectional_lockless/1048576` | corosio (IOCP) | 4.37 GB/s | 4.83% | +1.7% +| `unidirectional_lockless/1048576` | asio (coroutines) | 4.29 GB/s | 3.21% | -0.1% +| `unidirectional_lockless/1048576` | asio (callbacks) | 4.30 GB/s | 7.83% | baseline +| `unidirectional_lockless/16384` | corosio (IOCP) | 3.78 GB/s | 0.69% | -2.7% +| `unidirectional_lockless/16384` | asio (coroutines) | 3.72 GB/s | 0.87% | -4.1% +| `unidirectional_lockless/16384` | asio (callbacks) | 3.88 GB/s | 0.75% | baseline +| `unidirectional_lockless/262144` | corosio (IOCP) | 9.77 GB/s | 1.43% | -0.7% +| `unidirectional_lockless/262144` | asio (coroutines) | 9.76 GB/s | 0.85% | -0.8% +| `unidirectional_lockless/262144` | asio (callbacks) | 9.84 GB/s | 1.23% | baseline +| `unidirectional_lockless/4096` | corosio (IOCP) | 1.09 GB/s | 0.35% | -2.4% +| `unidirectional_lockless/4096` | asio (coroutines) | 1.07 GB/s | 0.30% | -3.9% +| `unidirectional_lockless/4096` | asio (callbacks) | 1.12 GB/s | 0.24% | baseline +| `unidirectional_lockless/65536` | corosio (IOCP) | 9.35 GB/s | 2.30% | -1.8% +| `unidirectional_lockless/65536` | asio (coroutines) | 9.26 GB/s | 1.66% | -2.8% +| `unidirectional_lockless/65536` | asio (callbacks) | 9.52 GB/s | 0.82% | baseline +|=== +==== + +-- +