From dcaf3fbb80a618f676816c44fcaf5854799997a8 Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Mon, 28 Sep 2026 17:43:35 +0200 Subject: [PATCH 1/8] feat(adoption): monthly HTTP Archive adoption numbers for spec topics Adds a monthly job that queries the HTTP Archive's custom metrics in BigQuery and refreshes src/data/adoption.json, which the spec pages render as an "Adoption: X% of origins" line under the summary. - scripts/adoption/metrics.json: spec slug -> BigQuery JSONPath conditions (17 topics: well-known paths, llms.txt, robots.txt, AI-crawler rules) - scripts/adoption/fetch-adoption.mjs: single aggregation query over the newest httparchive.pages crawl table; npm run adoption - .github/workflows/adoption-monthly.yml: runs on the 18th, opens/updates a PR like the Plausible refresh job; auth via Workload Identity Federation (setup in scripts/adoption/README.md) - SpecLayout renders the line when data exists for the page's slug Query cost is one small aggregation per month, inside BigQuery's free tier. --- package-lock.json | 555 +++++++++++++++++++++++++ package.json | 6 +- scripts/adoption/README.md | 43 ++ scripts/adoption/fetch-adoption.mjs | 163 ++++++++ scripts/adoption/metrics.json | 215 ++++++++++ src/data/adoption.json | 8 + src/layouts/SpecLayout.astro | 31 ++ src/pages/spec/[category]/[slug].astro | 20 + 8 files changed, 1039 insertions(+), 2 deletions(-) create mode 100644 scripts/adoption/README.md create mode 100644 scripts/adoption/fetch-adoption.mjs create mode 100644 scripts/adoption/metrics.json create mode 100644 src/data/adoption.json diff --git a/package-lock.json b/package-lock.json index 2fb4bbb7..8544b109 100644 --- a/package-lock.json +++ b/package-lock.json @@ -11,6 +11,7 @@ "dependencies": { "@astrojs/mdx": "^8.0.1", "@astrojs/rss": "^4.0.19", + "@google-cloud/bigquery": "^8.0.0", "@tailwindcss/vite": "^4.3.3", "astro": "^7.3.3", "dompurify": "^3.4.14", @@ -1309,6 +1310,104 @@ "node": "^20.19.0 || ^22.13.0 || >=24" } }, + "node_modules/@google-cloud/bigquery": { + "version": "8.3.1", + "resolved": "https://registry.npmjs.org/@google-cloud/bigquery/-/bigquery-8.3.1.tgz", + "integrity": "sha512-F4g9oMgI3EB5Uo+6npRHDSWn1HkVko8GebG1tJtzasn/gknAEFeobQzuLeGYC6SJ92RVdaBpGVisw86P70RrHQ==", + "license": "Apache-2.0", + "dependencies": { + "@google-cloud/common": "^6.0.0", + "@google-cloud/paginator": "^6.0.0", + "@google-cloud/precise-date": "^5.0.0", + "@google-cloud/promisify": "^5.0.0", + "arrify": "^3.0.0", + "big.js": "^7.0.0", + "duplexify": "^4.1.3", + "extend": "^3.0.2", + "stream-events": "^1.0.5", + "teeny-request": "^10.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@google-cloud/common": { + "version": "6.1.0", + "resolved": "https://registry.npmjs.org/@google-cloud/common/-/common-6.1.0.tgz", + "integrity": "sha512-Ohjxjvusr65+SiEVBrilpyn1ir9CM8ZqWjcf7bA8ph1GCunxI6bbwtcl7iGl67A7Lr4Nz7WpNzITNETwggc7pw==", + "license": "Apache-2.0", + "dependencies": { + "@google-cloud/projectify": "^4.0.0", + "@google-cloud/promisify": "^4.0.0", + "arrify": "^2.0.0", + "duplexify": "^4.1.3", + "extend": "^3.0.2", + "google-auth-library": "^10.6.2", + "html-entities": "^2.5.2", + "retry-request": "^8.0.0", + "teeny-request": "^10.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@google-cloud/common/node_modules/@google-cloud/promisify": { + "version": "4.1.0", + "resolved": "https://registry.npmjs.org/@google-cloud/promisify/-/promisify-4.1.0.tgz", + "integrity": "sha512-G/FQx5cE/+DqBbOpA5jKsegGwdPniU6PuIEMt+qxWgFxvxuFOzVmp6zYchtYuwAWV5/8Dgs0yAmjvNZv3uXLQg==", + "license": "Apache-2.0", + "engines": { + "node": ">=18" + } + }, + "node_modules/@google-cloud/common/node_modules/arrify": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/arrify/-/arrify-2.0.1.tgz", + "integrity": "sha512-3duEwti880xqi4eAMN8AyR4a0ByT90zoYdLlevfrvU43vb0YZwZVfxOgxWrLXXXpyugL0hNZc9G6BiB5B3nUug==", + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/@google-cloud/paginator": { + "version": "6.1.0", + "resolved": "https://registry.npmjs.org/@google-cloud/paginator/-/paginator-6.1.0.tgz", + "integrity": "sha512-9bxQ/QNhcq5c6ra75khWjA5Q9UHWp0vQnhwesmqPpLLF5PxmxCIsadIA5ssLxKted4/iZ2RlRPkW/E8p68W8cg==", + "license": "Apache-2.0", + "dependencies": { + "extend": "^3.0.2" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@google-cloud/precise-date": { + "version": "5.2.0", + "resolved": "https://registry.npmjs.org/@google-cloud/precise-date/-/precise-date-5.2.0.tgz", + "integrity": "sha512-ckZ1elVT/JXbO4kxltiT8tumYnTXGIY7OFrDMAJyVWAte/rUYiCuH1YbaKLQrZpYDp522xgTfPqqW/VxfXFdgA==", + "license": "Apache-2.0", + "engines": { + "node": ">=18" + } + }, + "node_modules/@google-cloud/projectify": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/@google-cloud/projectify/-/projectify-4.0.0.tgz", + "integrity": "sha512-MmaX6HeSvyPbWGwFq7mXdo0uQZLGBYCwziiLIGq5JVX+/bdI3SAq6bP98trV5eTWfLuvsMcIC1YJOF2vfteLFA==", + "license": "Apache-2.0", + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/@google-cloud/promisify": { + "version": "5.1.0", + "resolved": "https://registry.npmjs.org/@google-cloud/promisify/-/promisify-5.1.0.tgz", + "integrity": "sha512-/j9zzWDxsgKg0hMuFBmLGI8ES9xj4DjXvQnDCKl0w5ed7WE3CGsgHmUoC35rvfBXshSW+5StQGnCUD+RBqITKA==", + "license": "Apache-2.0", + "engines": { + "node": ">=18" + } + }, "node_modules/@humanfs/core": { "version": "0.19.2", "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.2.tgz", @@ -1392,6 +1491,7 @@ "cpu": [ "arm64" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1414,6 +1514,7 @@ "cpu": [ "x64" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1433,6 +1534,7 @@ "version": "0.35.4", "resolved": "https://registry.npmjs.org/@img/sharp-freebsd-wasm32/-/sharp-freebsd-wasm32-0.35.4.tgz", "integrity": "sha512-lIsKw/BU+kjB4eZjxrYrZmwOJYi3Ajrv66iAlBmUPyKc3HpnloevB1g3wxGD9P/5BbQ1brBGl65VRRrCvQDEqA==", + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1455,6 +1557,7 @@ "cpu": [ "arm64" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1471,6 +1574,7 @@ "cpu": [ "x64" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1490,6 +1594,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1509,6 +1614,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1528,6 +1634,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1547,6 +1654,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1566,6 +1674,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1585,6 +1694,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1604,6 +1714,7 @@ "libc": [ "musl" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1623,6 +1734,7 @@ "libc": [ "musl" ], + "dev": true, "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1642,6 +1754,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1667,6 +1780,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1692,6 +1806,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1717,6 +1832,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1742,6 +1858,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1767,6 +1884,7 @@ "libc": [ "glibc" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1792,6 +1910,7 @@ "libc": [ "musl" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1817,6 +1936,7 @@ "libc": [ "musl" ], + "dev": true, "license": "Apache-2.0", "optional": true, "os": [ @@ -1836,6 +1956,7 @@ "version": "0.35.4", "resolved": "https://registry.npmjs.org/@img/sharp-wasm32/-/sharp-wasm32-0.35.4.tgz", "integrity": "sha512-zQnl4Kwp7Q6NHsENtU2T/00Zi+w3AQNwz3+UaTyVBy2FpXrzXzGjndpK61onhZjRtRpQXxCTeqw19bVyXOh7jA==", + "dev": true, "license": "Apache-2.0 AND LGPL-3.0-or-later AND MIT", "optional": true, "dependencies": { @@ -1852,6 +1973,7 @@ "version": "1.11.3", "resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.11.3.tgz", "integrity": "sha512-Xz4Tpyki7XyrpbUK1jR1AhdAdaXyhhY4lZ3neLodmhpuWfy2PAQN5B46sAiU4liOXGLkHypn/qU+jvfWSCYYLA==", + "dev": true, "license": "MIT", "optional": true, "dependencies": { @@ -1865,6 +1987,7 @@ "cpu": [ "wasm32" ], + "dev": true, "license": "Apache-2.0", "optional": true, "dependencies": { @@ -1884,6 +2007,7 @@ "cpu": [ "arm64" ], + "dev": true, "license": "Apache-2.0 AND LGPL-3.0-or-later", "optional": true, "os": [ @@ -1903,6 +2027,7 @@ "cpu": [ "ia32" ], + "dev": true, "license": "Apache-2.0 AND LGPL-3.0-or-later", "optional": true, "os": [ @@ -1922,6 +2047,7 @@ "cpu": [ "x64" ], + "dev": true, "license": "Apache-2.0 AND LGPL-3.0-or-later", "optional": true, "os": [ @@ -3287,6 +3413,15 @@ "acorn": "^6.0.0 || ^7.0.0 || ^8.0.0" } }, + "node_modules/agent-base": { + "version": "7.1.4", + "resolved": "https://registry.npmjs.org/agent-base/-/agent-base-7.1.4.tgz", + "integrity": "sha512-MnA+YT8fwfJPgBx3m60MNqakm30XOkyIoH1y6huTQvC0PwZG7ki8NacLBcrPbNoo8vEZy7Jpuk7+jMO+CUovTQ==", + "license": "MIT", + "engines": { + "node": ">= 14" + } + }, "node_modules/ajv": { "version": "6.15.0", "resolved": "https://registry.npmjs.org/ajv/-/ajv-6.15.0.tgz", @@ -3382,6 +3517,18 @@ "node": ">= 0.4" } }, + "node_modules/arrify": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/arrify/-/arrify-3.0.0.tgz", + "integrity": "sha512-tLkvA81vQG/XqE2mjDkGQHoOINtMHtysSnemrmoGe6PydDPMRbVugqyk4A6V/WDWEfm3l+0d8anA9r8cv/5Jaw==", + "license": "MIT", + "engines": { + "node": ">=12" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/astro": { "version": "7.3.3", "resolved": "https://registry.npmjs.org/astro/-/astro-7.3.3.tgz", @@ -3716,6 +3863,48 @@ "node": "18 || 20 || >=22" } }, + "node_modules/base64-js": { + "version": "1.5.1", + "resolved": "https://registry.npmjs.org/base64-js/-/base64-js-1.5.1.tgz", + "integrity": "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT" + }, + "node_modules/big.js": { + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/big.js/-/big.js-7.0.1.tgz", + "integrity": "sha512-iFgV784tD8kq4ccF1xtNMZnXeZzVuXWWM+ERFzKQjv+A5G9HC8CY3DuV45vgzFFcW+u2tIvmF95+AzWgs6BjCg==", + "license": "MIT", + "engines": { + "node": "*" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/bigjs" + } + }, + "node_modules/bignumber.js": { + "version": "9.3.1", + "resolved": "https://registry.npmjs.org/bignumber.js/-/bignumber.js-9.3.1.tgz", + "integrity": "sha512-Ko0uX15oIUS7wJ3Rb30Fs6SkVbLmPBAKdlm7q9+ak9bbIeFf0MwuBsQV6z7+X768/cHsfg+WlysDWJcmthjsjQ==", + "license": "MIT", + "engines": { + "node": "*" + } + }, "node_modules/boolbase": { "version": "1.0.0", "resolved": "https://registry.npmjs.org/boolbase/-/boolbase-1.0.0.tgz", @@ -3735,6 +3924,12 @@ "node": "20 || >=22" } }, + "node_modules/buffer-equal-constant-time": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/buffer-equal-constant-time/-/buffer-equal-constant-time-1.0.1.tgz", + "integrity": "sha512-zRpUiDwd/xk6ADqPMATG8vc9VPrkck7T07OIx0gnjmJAnHnTVXNQG3vfvWNuiZIkwu9KrKdA1iJKfsfTVxE6NA==", + "license": "BSD-3-Clause" + }, "node_modules/cacheable": { "version": "2.5.0", "resolved": "https://registry.npmjs.org/cacheable/-/cacheable-2.5.0.tgz", @@ -4022,6 +4217,15 @@ "integrity": "sha512-aylIc7Z9y4yzHYAJNuESG3hfhC+0Ibp/MAMiaOZgNv4pmEdFyfZhhhny4MNiAfWdBQ1RQ2mfDWmM1x8SvGyp8g==", "license": "CC0-1.0" }, + "node_modules/data-uri-to-buffer": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/data-uri-to-buffer/-/data-uri-to-buffer-4.0.1.tgz", + "integrity": "sha512-0R9ikRb668HB7QDxT1vkpuUBtqc53YyAwMwGeUFKRojY/NWKvdZ+9UYtRfGmhqNbRkTSVpMbmyhXipFFv2cb/A==", + "license": "MIT", + "engines": { + "node": ">= 12" + } + }, "node_modules/debug": { "version": "4.4.3", "resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz", @@ -4190,6 +4394,27 @@ "node": ">=4" } }, + "node_modules/duplexify": { + "version": "4.1.3", + "resolved": "https://registry.npmjs.org/duplexify/-/duplexify-4.1.3.tgz", + "integrity": "sha512-M3BmBhwJRZsSx38lZyhE53Csddgzl5R7xGJNk7CVddZD6CcmwMCH8J+7AprIrQKH7TonKxaCjcv27Qmf+sQ+oA==", + "license": "MIT", + "dependencies": { + "end-of-stream": "^1.4.1", + "inherits": "^2.0.3", + "readable-stream": "^3.1.1", + "stream-shift": "^1.0.2" + } + }, + "node_modules/ecdsa-sig-formatter": { + "version": "1.0.11", + "resolved": "https://registry.npmjs.org/ecdsa-sig-formatter/-/ecdsa-sig-formatter-1.0.11.tgz", + "integrity": "sha512-nagl3RYrbNv6kQkeJIpt6NJZy8twLB/2vtz6yN9Z4vRKHN4/QZJIEbqohALSgwKdnksuY3k5Addp5lg8sVoVcQ==", + "license": "Apache-2.0", + "dependencies": { + "safe-buffer": "^5.0.1" + } + }, "node_modules/emmet": { "version": "2.4.11", "resolved": "https://registry.npmjs.org/emmet/-/emmet-2.4.11.tgz", @@ -4214,6 +4439,15 @@ "dev": true, "license": "MIT" }, + "node_modules/end-of-stream": { + "version": "1.4.5", + "resolved": "https://registry.npmjs.org/end-of-stream/-/end-of-stream-1.4.5.tgz", + "integrity": "sha512-ooEGc6HP26xXq/N+GCGOT0JKCLDGrq2bQUZrQ7gyrJiZANJ/8YDTxTpQBXGMn+WbIQXNVpyWymm7KYVICQnyOg==", + "license": "MIT", + "dependencies": { + "once": "^1.4.0" + } + }, "node_modules/enhanced-resolve": { "version": "5.24.5", "resolved": "https://registry.npmjs.org/enhanced-resolve/-/enhanced-resolve-5.24.5.tgz", @@ -4633,6 +4867,29 @@ } } }, + "node_modules/fetch-blob": { + "version": "3.2.0", + "resolved": "https://registry.npmjs.org/fetch-blob/-/fetch-blob-3.2.0.tgz", + "integrity": "sha512-7yAQpD2UMJzLi1Dqv7qFYnPbaPx7ZfFK6PiIxQ4PfkGPyNyl2Ugx+a/umUonmKqjhM4DnfbMvdX6otXq83soQQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/jimmywarting" + }, + { + "type": "paypal", + "url": "https://paypal.me/jimmywarting" + } + ], + "license": "MIT", + "dependencies": { + "node-domexception": "^1.0.0", + "web-streams-polyfill": "^3.0.3" + }, + "engines": { + "node": "^12.20 || >= 14.13" + } + }, "node_modules/file-entry-cache": { "version": "11.1.5", "resolved": "https://registry.npmjs.org/file-entry-cache/-/file-entry-cache-11.1.5.tgz", @@ -4718,6 +4975,18 @@ "node": ">=20" } }, + "node_modules/formdata-polyfill": { + "version": "4.0.10", + "resolved": "https://registry.npmjs.org/formdata-polyfill/-/formdata-polyfill-4.0.10.tgz", + "integrity": "sha512-buewHzMvYL29jdeQTVILecSaZKnt/RJWjoZCF5OW60Z67/GmSLBkOFM7qh1PI3zFNtJbaZL5eQu1vLfazOwj4g==", + "license": "MIT", + "dependencies": { + "fetch-blob": "^3.1.2" + }, + "engines": { + "node": ">=12.20.0" + } + }, "node_modules/fsevents": { "version": "2.3.3", "resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.3.tgz", @@ -4732,6 +5001,34 @@ "node": "^8.16.0 || ^10.6.0 || >=11.0.0" } }, + "node_modules/gaxios": { + "version": "7.3.1", + "resolved": "https://registry.npmjs.org/gaxios/-/gaxios-7.3.1.tgz", + "integrity": "sha512-kB3rzJV7d9juLZh8/56QTXCwQfxyhdOMdyYk1HdQKFtF8TJTDTZQJtixWIwXdE9Jji91mC41DUNpjleo4L4eAQ==", + "license": "Apache-2.0", + "dependencies": { + "extend": "^3.0.2", + "https-proxy-agent": "^7.0.1", + "node-fetch": "^3.3.2" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/gcp-metadata": { + "version": "8.1.2", + "resolved": "https://registry.npmjs.org/gcp-metadata/-/gcp-metadata-8.1.2.tgz", + "integrity": "sha512-zV/5HKTfCeKWnxG0Dmrw51hEWFGfcF2xiXqcA3+J90WDuP0SvoiSO5ORvcBsifmx/FoIjgQN3oNOGaQ5PhLFkg==", + "license": "Apache-2.0", + "dependencies": { + "gaxios": "^7.0.0", + "google-logging-utils": "^1.0.0", + "json-bigint": "^1.0.0" + }, + "engines": { + "node": ">=18" + } + }, "node_modules/get-caller-file": { "version": "2.0.5", "resolved": "https://registry.npmjs.org/get-caller-file/-/get-caller-file-2.0.5.tgz", @@ -4820,6 +5117,32 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/google-auth-library": { + "version": "10.9.1", + "resolved": "https://registry.npmjs.org/google-auth-library/-/google-auth-library-10.9.1.tgz", + "integrity": "sha512-i1ydyHrqcIxXkWh/uBmVkzCvIuq5yiK2ATndIe5XxKholrG/MTYP9xGYka4sQhrbIAgGjL2B6NOE7rFaiF3fXw==", + "license": "Apache-2.0", + "dependencies": { + "base64-js": "^1.3.0", + "ecdsa-sig-formatter": "^1.0.11", + "gaxios": "^7.1.4", + "gcp-metadata": "8.1.2", + "google-logging-utils": "1.1.3", + "jws": "^4.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/google-logging-utils": { + "version": "1.1.3", + "resolved": "https://registry.npmjs.org/google-logging-utils/-/google-logging-utils-1.1.3.tgz", + "integrity": "sha512-eAmLkjDjAFCVXg7A1unxHsLf961m6y17QFqXqAXGj/gVkKFrEICfStRfwUlGNfeCEjNRa32JEWOUTlYXPyyKvA==", + "license": "Apache-2.0", + "engines": { + "node": ">=14" + } + }, "node_modules/graceful-fs": { "version": "4.2.11", "resolved": "https://registry.npmjs.org/graceful-fs/-/graceful-fs-4.2.11.tgz", @@ -4899,6 +5222,22 @@ "dev": true, "license": "MIT" }, + "node_modules/html-entities": { + "version": "2.6.0", + "resolved": "https://registry.npmjs.org/html-entities/-/html-entities-2.6.0.tgz", + "integrity": "sha512-kig+rMn/QOVRvr7c86gQ8lWXq+Hkv6CbAH1hLu+RG338StTpE8Z0b44SDVaqVu7HGKf27frdmUYEs9hTUX/cLQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/mdevils" + }, + { + "type": "patreon", + "url": "https://patreon.com/mdevils" + } + ], + "license": "MIT" + }, "node_modules/html-escaper": { "version": "3.0.3", "resolved": "https://registry.npmjs.org/html-escaper/-/html-escaper-3.0.3.tgz", @@ -5028,6 +5367,32 @@ "integrity": "sha512-dTxcvPXqPvXBQpq5dUr6mEMJX4oIEFv6bwom3FDwKRDsuIjjJGANqhBuoAn9c1RQJIdAKav33ED65E2ys+87QQ==", "license": "BSD-2-Clause" }, + "node_modules/http-proxy-agent": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/http-proxy-agent/-/http-proxy-agent-7.0.2.tgz", + "integrity": "sha512-T1gkAiYYDWYx3V5Bmyu7HcfcvL7mUrTWiM6yOfa3PIphViJ/gFPbvidQ+veqSOHci/PxBcDabeUNCzpOODJZig==", + "license": "MIT", + "dependencies": { + "agent-base": "^7.1.0", + "debug": "^4.3.4" + }, + "engines": { + "node": ">= 14" + } + }, + "node_modules/https-proxy-agent": { + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/https-proxy-agent/-/https-proxy-agent-7.0.6.tgz", + "integrity": "sha512-vK9P5/iUfdl95AI+JVyUuIcVtd4ofvtrOr3HNtM2yxC9bnMbEdp3x01OhQNnjb8IJYi38VlTE3mBXwcfvywuSw==", + "license": "MIT", + "dependencies": { + "agent-base": "^7.1.2", + "debug": "4" + }, + "engines": { + "node": ">= 14" + } + }, "node_modules/ignore": { "version": "5.3.2", "resolved": "https://registry.npmjs.org/ignore/-/ignore-5.3.2.tgz", @@ -5048,6 +5413,12 @@ "node": ">=0.8.19" } }, + "node_modules/inherits": { + "version": "2.0.4", + "resolved": "https://registry.npmjs.org/inherits/-/inherits-2.0.4.tgz", + "integrity": "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ==", + "license": "ISC" + }, "node_modules/iron-webcrypto": { "version": "1.2.1", "resolved": "https://registry.npmjs.org/iron-webcrypto/-/iron-webcrypto-1.2.1.tgz", @@ -5145,6 +5516,15 @@ "js-yaml": "bin/js-yaml.js" } }, + "node_modules/json-bigint": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/json-bigint/-/json-bigint-1.0.0.tgz", + "integrity": "sha512-SiPv/8VpZuWbvLSMtTDU8hEfrZWg/mH/nV/b4o0CYbSxu1UIQPLdwKOCIyLQX+VIPO5vrLX3i8qtqFyhdPSUSQ==", + "license": "MIT", + "dependencies": { + "bignumber.js": "^9.0.0" + } + }, "node_modules/json-schema-traverse": { "version": "0.4.1", "resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-0.4.1.tgz", @@ -5165,6 +5545,27 @@ "integrity": "sha512-HUgH65KyejrUFPvHFPbqOY0rsFip3Bo5wb4ngvdi1EpCYWUQDC5V+Y7mZws+DLkr4M//zQJoanu1SP+87Dv1oQ==", "license": "MIT" }, + "node_modules/jwa": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/jwa/-/jwa-2.0.1.tgz", + "integrity": "sha512-hRF04fqJIP8Abbkq5NKGN0Bbr3JxlQ+qhZufXVr0DvujKy93ZCbXZMHDL4EOtodSbCWxOqR8MS1tXA5hwqCXDg==", + "license": "MIT", + "dependencies": { + "buffer-equal-constant-time": "^1.0.1", + "ecdsa-sig-formatter": "1.0.11", + "safe-buffer": "^5.0.1" + } + }, + "node_modules/jws": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/jws/-/jws-4.0.1.tgz", + "integrity": "sha512-EKI/M/yqPncGUUh44xz0PxSidXFr/+r0pA70+gIYhjv+et7yxM+s29Y+VGDkovRofQem0fs7Uvf4+YmAdyRduA==", + "license": "MIT", + "dependencies": { + "jwa": "^2.0.1", + "safe-buffer": "^5.0.1" + } + }, "node_modules/keyv": { "version": "5.6.0", "resolved": "https://registry.npmjs.org/keyv/-/keyv-5.6.0.tgz", @@ -5786,6 +6187,44 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/node-domexception": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/node-domexception/-/node-domexception-1.0.0.tgz", + "integrity": "sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ==", + "deprecated": "Use your platform's native DOMException instead", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/jimmywarting" + }, + { + "type": "github", + "url": "https://paypal.me/jimmywarting" + } + ], + "license": "MIT", + "engines": { + "node": ">=10.5.0" + } + }, + "node_modules/node-fetch": { + "version": "3.3.2", + "resolved": "https://registry.npmjs.org/node-fetch/-/node-fetch-3.3.2.tgz", + "integrity": "sha512-dRB78srN/l6gqWulah9SrxeYnxeddIG30+GOqK/9OlLVyLg3HPnr6SqOWTWOXKRwC2eGYCkZ59NNuSgvSrpgOA==", + "license": "MIT", + "dependencies": { + "data-uri-to-buffer": "^4.0.0", + "fetch-blob": "^3.1.4", + "formdata-polyfill": "^4.0.10" + }, + "engines": { + "node": "^12.20.0 || ^14.13.1 || >=16.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/node-fetch" + } + }, "node_modules/node-fetch-native": { "version": "1.6.7", "resolved": "https://registry.npmjs.org/node-fetch-native/-/node-fetch-native-1.6.7.tgz", @@ -5846,6 +6285,15 @@ "integrity": "sha512-65S/5gk9YSsaRjcyf7Nfa6h/d3E8/1gslpXfI4W7Dxn/oap8IKRuNT5VXkLQ1YFKIEg4apRY4Pj6aiwFzrDdmw==", "license": "MIT" }, + "node_modules/once": { + "version": "1.4.0", + "resolved": "https://registry.npmjs.org/once/-/once-1.4.0.tgz", + "integrity": "sha512-lNaJgI+2Q5URQBkccEKHTQOPaXdUxnZZElQTZY0MFUAuaEqe1E+Nyvgdz/aIyNi6Z9MzO5dv1H8n58/GELp3+w==", + "license": "ISC", + "dependencies": { + "wrappy": "1" + } + }, "node_modules/oniguruma-parser": { "version": "0.12.2", "resolved": "https://registry.npmjs.org/oniguruma-parser/-/oniguruma-parser-0.12.2.tgz", @@ -6426,6 +6874,20 @@ "integrity": "sha512-b484I/7b8rDEdSDKckSSBA8knMpcdsXudlE/LNL639wFoHKwLbEkQFZHWEYwDC0wa0FKUcCY+GAF73Z7wxNVFA==", "license": "MIT" }, + "node_modules/readable-stream": { + "version": "3.6.2", + "resolved": "https://registry.npmjs.org/readable-stream/-/readable-stream-3.6.2.tgz", + "integrity": "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA==", + "license": "MIT", + "dependencies": { + "inherits": "^2.0.3", + "string_decoder": "^1.1.1", + "util-deprecate": "^1.0.1" + }, + "engines": { + "node": ">= 6" + } + }, "node_modules/readdirp": { "version": "5.0.0", "resolved": "https://registry.npmjs.org/readdirp/-/readdirp-5.0.0.tgz", @@ -6504,6 +6966,19 @@ "url": "https://opencollective.com/unified" } }, + "node_modules/retry-request": { + "version": "8.0.4", + "resolved": "https://registry.npmjs.org/retry-request/-/retry-request-8.0.4.tgz", + "integrity": "sha512-pI6/7eabUYkZxamkOq0g0uMxKLLGnjzhefY+vL8bVXag5rto4OU2YBTPytWLuHH7aEKD6fn7kJycQfid0Mwnkw==", + "license": "MIT", + "dependencies": { + "extend": "^3.0.2", + "teeny-request": "^10.0.0" + }, + "engines": { + "node": ">=18" + } + }, "node_modules/rolldown": { "version": "1.1.3", "resolved": "https://registry.npmjs.org/rolldown/-/rolldown-1.1.3.tgz", @@ -6544,6 +7019,26 @@ "dev": true, "license": "MIT" }, + "node_modules/safe-buffer": { + "version": "5.2.1", + "resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz", + "integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT" + }, "node_modules/sass-formatter": { "version": "0.8.0", "resolved": "https://registry.npmjs.org/sass-formatter/-/sass-formatter-0.8.0.tgz", @@ -6757,6 +7252,30 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/stream-events": { + "version": "1.0.5", + "resolved": "https://registry.npmjs.org/stream-events/-/stream-events-1.0.5.tgz", + "integrity": "sha512-E1GUzBSgvct8Jsb3v2X15pjzN1tYebtbLaMg+eBOUOAxgbLoSbT2NS91ckc5lJD1KfLjId+jXJRgo0qnV5Nerg==", + "license": "MIT", + "dependencies": { + "stubs": "^3.0.0" + } + }, + "node_modules/stream-shift": { + "version": "1.0.3", + "resolved": "https://registry.npmjs.org/stream-shift/-/stream-shift-1.0.3.tgz", + "integrity": "sha512-76ORR0DO1o1hlKwTbi/DM3EXWGf3ZJYO8cXX5RJwnul2DEg2oyoZyjLNoQM8WsvZiFKCRfC1O0J7iCvie3RZmQ==", + "license": "MIT" + }, + "node_modules/string_decoder": { + "version": "1.3.0", + "resolved": "https://registry.npmjs.org/string_decoder/-/string_decoder-1.3.0.tgz", + "integrity": "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==", + "license": "MIT", + "dependencies": { + "safe-buffer": "~5.2.0" + } + }, "node_modules/string-width": { "version": "8.2.2", "resolved": "https://registry.npmjs.org/string-width/-/string-width-8.2.2.tgz", @@ -6816,6 +7335,12 @@ ], "license": "MIT" }, + "node_modules/stubs": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/stubs/-/stubs-3.0.0.tgz", + "integrity": "sha512-PdHt7hHUJKxvTCgbKX9C1V/ftOcjJQgz8BZwNfV5c4B6dcGqlpelTbJ999jBGZ2jYiPAwcX5dP6oBwVlBlUbxw==", + "license": "MIT" + }, "node_modules/suf-log": { "version": "2.5.3", "resolved": "https://registry.npmjs.org/suf-log/-/suf-log-2.5.3.tgz", @@ -6886,6 +7411,21 @@ "url": "https://opencollective.com/webpack" } }, + "node_modules/teeny-request": { + "version": "10.1.4", + "resolved": "https://registry.npmjs.org/teeny-request/-/teeny-request-10.1.4.tgz", + "integrity": "sha512-R1Cg4Vu0UULeDfHL/kjABLaTW++9yD/B6n2g48y5dJ04hsEaxcfmAqbNDzNsbqAYJyIpZafjklLG9YxRu9uzOg==", + "license": "Apache-2.0", + "dependencies": { + "http-proxy-agent": "^7.0.0", + "https-proxy-agent": "^7.0.1", + "node-fetch": "^3.3.2", + "stream-events": "^1.0.5" + }, + "engines": { + "node": ">=18" + } + }, "node_modules/tiny-inflate": { "version": "1.0.3", "resolved": "https://registry.npmjs.org/tiny-inflate/-/tiny-inflate-1.0.3.tgz", @@ -7698,6 +8238,15 @@ "dev": true, "license": "MIT" }, + "node_modules/web-streams-polyfill": { + "version": "3.3.3", + "resolved": "https://registry.npmjs.org/web-streams-polyfill/-/web-streams-polyfill-3.3.3.tgz", + "integrity": "sha512-d2JWLCivmZYTSIoge9MsgFCZrt571BikcWGYkjC1khllbTeDlGqZ2D8vD8E/lJa8WGWbb7Plm8/XJYV7IJHZZw==", + "license": "MIT", + "engines": { + "node": ">= 8" + } + }, "node_modules/which": { "version": "2.0.2", "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", @@ -7760,6 +8309,12 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/wrappy": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/wrappy/-/wrappy-1.0.2.tgz", + "integrity": "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==", + "license": "ISC" + }, "node_modules/xml-naming": { "version": "0.1.0", "resolved": "https://registry.npmjs.org/xml-naming/-/xml-naming-0.1.0.tgz", diff --git a/package.json b/package.json index 0c757a85..9a5aad9f 100644 --- a/package.json +++ b/package.json @@ -30,7 +30,8 @@ "assets": "node scripts/generate-assets.mjs", "test:websub": "node --experimental-strip-types scripts/test-websub.mjs", "sign:skill": "node scripts/check-skill.mjs", - "check:skill": "node scripts/check-skill.mjs --check" + "check:skill": "node scripts/check-skill.mjs --check", + "adoption": "node scripts/adoption/fetch-adoption.mjs" }, "dependencies": { "@astrojs/mdx": "^8.0.1", @@ -38,7 +39,8 @@ "@tailwindcss/vite": "^4.3.3", "astro": "^7.3.3", "dompurify": "^3.4.14", - "tailwindcss": "^4.3.0" + "tailwindcss": "^4.3.0", + "@google-cloud/bigquery": "^8.0.0" }, "devDependencies": { "@astrojs/check": "^0.9.10", diff --git a/scripts/adoption/README.md b/scripts/adoption/README.md new file mode 100644 index 00000000..7d326d4c --- /dev/null +++ b/scripts/adoption/README.md @@ -0,0 +1,43 @@ +# HTTP Archive adoption data + +Monthly adoption percentages for spec topics, pulled from the HTTP Archive's +custom metrics via BigQuery. The numbers land in `src/data/adoption.json` and +the spec pages render them automatically. + +## One-time setup (Google Cloud) + +The HTTP Archive dataset is public, but BigQuery bills the _querying_ project, +so you need a small GCP project of your own. One aggregation query per month +fits comfortably inside the 1 TiB/month free tier. + +1. Create a GCP project and enable the BigQuery API. +2. Create a service account (no keys needed) with the **BigQuery Job User** + role on the project. That role is enough: it grants `bigquery.jobs.create` + for billing; the `httparchive` dataset is public. +3. Set up Workload Identity Federation for GitHub Actions (no long-lived + secrets): create a workload identity pool + OIDC provider trusting + `https://token.actions.githubusercontent.com`, allow the service account to + be impersonated by this repo (`attribute.repository = jdevalk/specification.website`). + The [official guide](https://docs.github.com/en/actions/how-tos/secure-your-work/deployments/oidc-in-gcp) + walks through it. +4. Set these **repository variables** (Settings → Variables → Actions): + - `GCP_PROJECT_ID` — your project id + - `GCP_WIF_PROVIDER` — the full provider resource name + - `GCP_SERVICE_ACCOUNT` — the service account email + +## Running it + +- Automatically: `.github/workflows/adoption-monthly.yml` runs on the 18th of + each month (crawl data lands mid-month) and opens or updates a PR with the + refreshed `src/data/adoption.json`. +- Manually: `npm run adoption` (needs `GCP_PROJECT_ID` set and Application + Default Credentials, e.g. via `gcloud auth application-default login`). +- The workflow can also be triggered by hand via workflow_dispatch. + +## Adding a topic + +Add an entry to `metrics.json`: the spec page `slug` plus one or more +conditions against the `custom_metrics` JSON column. Find the exact key and +field shapes in the [custom-metrics repo](https://github.com/HTTPArchive/custom-metrics/tree/main/dist). +The script warns when a configured top-level key is missing from the crawl, so +a typo surfaces on the first run instead of silently reading 0. diff --git a/scripts/adoption/fetch-adoption.mjs b/scripts/adoption/fetch-adoption.mjs new file mode 100644 index 00000000..395ca2a9 --- /dev/null +++ b/scripts/adoption/fetch-adoption.mjs @@ -0,0 +1,163 @@ +#!/usr/bin/env node +/** + * Fetches HTTP Archive custom-metric adoption numbers for spec topics and + * writes them to src/data/adoption.json, which the spec pages render. + * + * Reads its metric definitions from ./metrics.json. Each metric becomes one + * COUNTIF over the latest `httparchive.pages.YYYY_MM_01_desktop` table, so a + * full refresh is a single BigQuery aggregation query. + * + * Auth: Application Default Credentials. Locally that means + * `gcloud auth application-default login`; in CI the adoption workflow + * authenticates via Workload Identity Federation. Query billing lands on + * GCP_PROJECT_ID (the httparchive dataset itself is public). + * + * Usage: GCP_PROJECT_ID=your-project node scripts/adoption/fetch-adoption.mjs + */ +import { readFileSync, writeFileSync } from "node:fs"; +import { dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { BigQuery } from "@google-cloud/bigquery"; + +const here = dirname(fileURLToPath(import.meta.url)); +const projectId = process.env.GCP_PROJECT_ID; +if (!projectId) { + console.error( + "error: GCP_PROJECT_ID is required (project billed for the query)", + ); + process.exit(1); +} + +const config = JSON.parse(readFileSync(resolve(here, "metrics.json"), "utf8")); +const bigquery = new BigQuery({ projectId }); + +/** Newest httparchive.pages crawl table that actually exists (data lands mid-month). */ +async function latestCrawlTable() { + const now = new Date(); + for (let back = 0; back <= 6; back++) { + const d = new Date( + Date.UTC(now.getUTCFullYear(), now.getUTCMonth() - back, 1), + ); + const suffix = `${d.getUTCFullYear()}_${String(d.getUTCMonth() + 1).padStart(2, "0")}_01_desktop`; + try { + await bigquery.query({ + query: `SELECT 1 FROM \`httparchive.pages.${suffix}\` LIMIT 0`, + dryRun: false, + }); + return suffix; + } catch (error) { + if (error?.code !== 404) throw error; + } + } + throw new Error( + "no httparchive.pages crawl table found in the last 6 months", + ); +} + +function conditionSql(condition) { + if (condition.op === "eq") { + return `JSON_VALUE(custom_metrics, '${condition.jsonPath}') = '${condition.value}'`; + } + if (condition.op === "exists") { + return `JSON_QUERY(custom_metrics, '${condition.jsonPath}') IS NOT NULL`; + } + throw new Error(`unknown condition op: ${condition.op}`); +} + +function metricSql(metric, index) { + const parts = metric.conditions.map(conditionSql); + const combined = + metric.combine === "all" ? parts.join(" AND ") : parts.join(" OR "); + return `COUNTIF(${combined}) AS pages_m${index}`; +} + +/** Warn when a configured top-level custom-metric key is absent from the crawl. */ +async function checkKeys(table, keys) { + const [rows] = await bigquery.query({ + query: `SELECT custom_metrics FROM \`httparchive.pages.${table}\` LIMIT 1`, + }); + const parsed = JSON.parse(rows[0]?.custom_metrics ?? "{}"); + const missing = [...new Set(keys)].filter((k) => !(k in parsed)); + for (const key of missing) { + console.warn( + `warning: custom metric key "${key}" not present in crawl ${table}; ` + + `those adoption numbers will read 0. Available keys: ${Object.keys(parsed).slice(0, 12).join(", ")}…`, + ); + } +} + +/** Origins-based COUNTIF needs the host per hit; do it with a second pass over hosts. */ +function originsSql(metric, index) { + const parts = metric.conditions.map(conditionSql); + const combined = + metric.combine === "all" ? parts.join(" AND ") : parts.join(" OR "); + return `COUNT(DISTINCT IF(${combined}, NET.HOST(url), NULL)) AS origins_m${index}`; +} + +function buildQuery(table) { + const selects = config.metrics + .flatMap((metric, i) => [metricSql(metric, i), originsSql(metric, i)]) + .join(",\n "); + return `SELECT + COUNT(*) AS pages, + COUNT(DISTINCT NET.HOST(url)) AS origins, + ${selects} + FROM \`httparchive.pages.${table}\``; +} + +async function main() { + // Offline mode: print the generated SQL without touching BigQuery. + if (process.env.ADOPTION_PRINT_QUERY) { + console.log(buildQuery("2026_09_01_desktop")); + return; + } + + const table = await latestCrawlTable(); + const crawl = table.slice(0, 7).replace("_", "-"); + console.log(`using crawl table httparchive.pages.${table}`); + + const topKeys = config.metrics.flatMap((m) => + m.conditions.map((c) => c.jsonPath.split(".")[1]), + ); + await checkKeys(table, topKeys); + + const query = buildQuery(table); + const [rows] = await bigquery.query({ query }); + const row = rows[0]; + + const metrics = {}; + config.metrics.forEach((metric, i) => { + const pages = Number(row[`pages_m${i}`]); + const origins = Number(row[`origins_m${i}`]); + metrics[metric.slug] = { + label: metric.label, + pages, + pagesPct: roundPct(pages / Number(row.pages)), + origins, + originsPct: roundPct(origins / Number(row.origins)), + }; + }); + + const out = { + crawl, + generatedAt: new Date().toISOString(), + source: "HTTP Archive monthly crawl (desktop), custom metrics", + pages: Number(row.pages), + origins: Number(row.origins), + metrics, + }; + const dest = resolve(here, "../../src/data/adoption.json"); + writeFileSync(dest, JSON.stringify(out, null, 2) + "\n"); + console.log( + `wrote ${dest} (${config.metrics.length} metrics, crawl ${crawl})`, + ); +} + +function roundPct(ratio) { + return Math.round(ratio * 100 * 1000) / 1000; +} + +main().catch((error) => { + console.error(`error: ${error.message}`); + process.exit(1); +}); diff --git a/scripts/adoption/metrics.json b/scripts/adoption/metrics.json new file mode 100644 index 00000000..7290fe4e --- /dev/null +++ b/scripts/adoption/metrics.json @@ -0,0 +1,215 @@ +{ + "$comment": "Maps spec page slugs to HTTP Archive custom-metric signals. Each metric lists one or more JSONPath conditions against the `custom_metrics` column of `httparchive.pages.YYYY_MM_01_desktop`. The fetch script builds one COUNTIF per metric; `combine: any` counts a page when at least one condition holds.", + "metrics": [ + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/gpc.json\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/gpc.json", + "slug": "gpc-json" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/security.txt\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/security.txt", + "slug": "security-txt" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/assetlinks.json\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/assetlinks.json", + "slug": "assetlinks-json" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/apple-app-site-association\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/apple-app-site-association", + "slug": "apple-app-site-association" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/change-password\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/change-password", + "slug": "change-password" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/webauthn\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/webauthn", + "slug": "webauthn" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/ai-catalog.json\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/ai-catalog.json (agent catalog)", + "slug": "api-catalog" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/ard.json\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/ard.json", + "note": "Measured from the October 2026 crawl onward (metric added September 2026).", + "slug": "agentic-resource-discovery" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/nodeinfo\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/nodeinfo", + "note": "Needs the well-known metric extension to land upstream first.", + "slug": "nodeinfo" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/webfinger\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/webfinger", + "note": "Needs the well-known metric extension to land upstream first.", + "slug": "webfinger" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/oauth-authorization-server\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/oauth-authorization-server", + "note": "Needs the well-known metric extension to land upstream first.", + "slug": "oauth-authorization-server" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/oauth-protected-resource\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/oauth-protected-resource", + "note": "Needs the well-known metric extension to land upstream first.", + "slug": "oauth-protected-resource" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/openid-configuration\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/openid-configuration", + "note": "Needs the well-known metric extension to land upstream first.", + "slug": "openid-configuration" + }, + { + "conditions": [ + { + "jsonPath": "$._well-known.\"/.well-known/traffic-advice\".found", + "op": "eq", + "value": "true" + } + ], + "label": "/.well-known/traffic-advice", + "note": "Needs the well-known metric extension to land upstream first.", + "slug": "traffic-advice" + }, + { + "conditions": [ + { "jsonPath": "$._llms-txt-valid.valid", "op": "eq", "value": "true" } + ], + "label": "/llms.txt (valid)", + "slug": "llms-txt" + }, + { + "conditions": [ + { "jsonPath": "$._robots_txt.status", "op": "eq", "value": "200" } + ], + "label": "/robots.txt (HTTP 200)", + "slug": "robots-txt" + }, + { + "combine": "any", + "conditions": [ + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"gptbot\"", + "op": "exists" + }, + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"claudebot\"", + "op": "exists" + }, + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"ccbot\"", + "op": "exists" + }, + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"anthropic-ai\"", + "op": "exists" + }, + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"google-extended\"", + "op": "exists" + }, + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"perplexitybot\"", + "op": "exists" + }, + { + "jsonPath": "$._robots_txt.record_counts.by_useragent.\"bytespider\"", + "op": "exists" + } + ], + "label": "robots.txt rules naming AI crawlers", + "slug": "robots-for-ai-crawlers" + } + ] +} diff --git a/src/data/adoption.json b/src/data/adoption.json new file mode 100644 index 00000000..d77247c4 --- /dev/null +++ b/src/data/adoption.json @@ -0,0 +1,8 @@ +{ + "crawl": null, + "generatedAt": null, + "metrics": {}, + "origins": 0, + "pages": 0, + "source": "HTTP Archive monthly crawl (desktop), custom metrics" +} diff --git a/src/layouts/SpecLayout.astro b/src/layouts/SpecLayout.astro index 77e462e8..09791cce 100644 --- a/src/layouts/SpecLayout.astro +++ b/src/layouts/SpecLayout.astro @@ -20,6 +20,14 @@ interface Props { updated?: string; appliesTo?: string[]; slug: string; + adoption?: { + label: string; + pages: number; + pagesPct: number; + origins: number; + originsPct: number; + } | null; + adoptionCrawl?: string | null; } const { @@ -32,8 +40,24 @@ const { updated, appliesTo = ["all"], slug, + adoption = null, + adoptionCrawl = null, } = Astro.props; +const adoptionLabel = + !adoption || adoption.origins === 0 + ? null + : adoption.originsPct > 0 + ? `${adoption.originsPct}%` + : "<0.001%"; +const crawlLabel = adoptionCrawl + ? new Date(`${adoptionCrawl}-02T00:00:00Z`).toLocaleString("en-GB", { + month: "long", + year: "numeric", + timeZone: "UTC", + }) + : null; + const cat = categoryFor(category); const canonical = new URL(`/spec/${category}/${slug}/`, site.url).toString(); const markdownUrl = new URL( @@ -133,6 +157,13 @@ const breadcrumbJsonLd = {

{summary}

+ {adoptionLabel && crawlLabel && ( +

+ Adoption:{" "} + {adoptionLabel} of + origins in the HTTP Archive {crawlLabel} crawl +

+ )} {appliesTo.length > 0 && appliesTo[0] !== "all" && (

Applies to:{" "} diff --git a/src/pages/spec/[category]/[slug].astro b/src/pages/spec/[category]/[slug].astro index de5c7689..c81ef0ae 100644 --- a/src/pages/spec/[category]/[slug].astro +++ b/src/pages/spec/[category]/[slug].astro @@ -1,6 +1,23 @@ --- import { getCollection, render } from "astro:content"; import SpecLayout from "~/layouts/SpecLayout.astro"; +import adoptionDataRaw from "~/data/adoption.json"; + +interface AdoptionMetric { + label: string; + pages: number; + pagesPct: number; + origins: number; + originsPct: number; +} + +interface AdoptionData { + crawl: string | null; + generatedAt: string | null; + metrics: Record; +} + +const adoptionData = adoptionDataRaw as AdoptionData; export async function getStaticPaths() { const all = await getCollection("spec", ({ data }) => !data.draft); @@ -32,6 +49,7 @@ const related = entry.data.relatedSlugs .filter((r): r is NonNullable => r !== null); const slug = entry.data.slug ?? entry.id.split("/").pop()!; +const adoption = adoptionData.metrics[slug] ?? null; --- From 7119192a1a0e715cc83fad15a67e3c8039fe7d00 Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Mon, 28 Sep 2026 22:18:12 +0200 Subject: [PATCH 2/8] ci: add monthly HTTP Archive adoption refresh --- .github/workflows/adoption-monthly.yml | 85 ++++++++++++++++++++++++++ 1 file changed, 85 insertions(+) create mode 100644 .github/workflows/adoption-monthly.yml diff --git a/.github/workflows/adoption-monthly.yml b/.github/workflows/adoption-monthly.yml new file mode 100644 index 00000000..e4ac78b2 --- /dev/null +++ b/.github/workflows/adoption-monthly.yml @@ -0,0 +1,85 @@ +name: Adoption refresh + +# Pulls HTTP Archive custom-metric adoption numbers for spec topics +# (scripts/adoption/) and opens (or updates) a single PR refreshing +# src/data/adoption.json, which the spec pages render automatically. +# +# Runs monthly on the 18th: the crawl for month M lands in BigQuery around +# mid-month M+1, so the 18th reliably picks up the newest complete crawl. +# The fetch script itself also falls back to the newest table it can find. +# +# One-time setup is documented in scripts/adoption/README.md (GCP project + +# Workload Identity Federation; no long-lived secrets). + +on: + schedule: + - cron: "0 6 18 * *" # 18th of each month, 06:00 UTC + workflow_dispatch: + +permissions: + contents: write + pull-requests: write + id-token: write + +jobs: + refresh: + name: Refresh adoption data + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + + - uses: actions/setup-node@v4 + with: + node-version: 22 + cache: npm + + - uses: google-github-actions/auth@v2 + with: + project_id: ${{ vars.GCP_PROJECT_ID }} + workload_identity_provider: ${{ vars.GCP_WIF_PROVIDER }} + service_account: ${{ vars.GCP_SERVICE_ACCOUNT }} + + - name: Install dependencies + run: npm ci + + - name: Fetch adoption data + run: npm run adoption + env: + GCP_PROJECT_ID: ${{ vars.GCP_PROJECT_ID }} + + - name: Format data file + run: npx prettier --write src/data/adoption.json + + - name: Detect change + id: diff + run: | + if git diff --quiet -- src/data/adoption.json; then + echo "changed=false" >> "$GITHUB_OUTPUT" + echo "Adoption data unchanged." + else + echo "changed=true" >> "$GITHUB_OUTPUT" + fi + + - name: Open or update PR + if: steps.diff.outputs.changed == 'true' + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + branch="chore/refresh-adoption" + crawl=$(node -p "JSON.parse(require('fs').readFileSync('src/data/adoption.json','utf8')).crawl") + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git checkout -B "$branch" + git add src/data/adoption.json + git commit -m "chore(adoption): refresh HTTP Archive adoption data ($crawl)" + git push -f -u origin "$branch" + body_file=$(mktemp) + trap 'rm -f "$body_file"' EXIT + printf 'Monthly refresh of the HTTP Archive adoption numbers rendered on the spec pages (`src/data/adoption.json`, crawl `%s`).\n\nGenerated by `npm run adoption` — a single BigQuery aggregation over the newest `httparchive.pages` crawl table. Merge to deploy.\n' "$crawl" > "$body_file" + if gh pr view "$branch" --json number >/dev/null 2>&1; then + gh pr edit "$branch" --body-file "$body_file" + else + gh pr create --base main --head "$branch" \ + --title "chore(adoption): refresh HTTP Archive adoption data ($crawl)" \ + --body-file "$body_file" + fi From 89713093868f4fe092ecd59b95e046e42469c87f Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Mon, 28 Sep 2026 22:24:33 +0200 Subject: [PATCH 3/8] fix(adoption): query current HTTP Archive schema and support dry runs --- .github/workflows/adoption-monthly.yml | 6 +- scripts/adoption/README.md | 34 +++- scripts/adoption/fetch-adoption.mjs | 230 +++++++++++++---------- scripts/adoption/fetch-adoption.test.mjs | 133 +++++++++++++ scripts/adoption/metrics.json | 121 +++++++----- 5 files changed, 366 insertions(+), 158 deletions(-) create mode 100644 scripts/adoption/fetch-adoption.test.mjs diff --git a/.github/workflows/adoption-monthly.yml b/.github/workflows/adoption-monthly.yml index e4ac78b2..cdee4be0 100644 --- a/.github/workflows/adoption-monthly.yml +++ b/.github/workflows/adoption-monthly.yml @@ -5,8 +5,8 @@ name: Adoption refresh # src/data/adoption.json, which the spec pages render automatically. # # Runs monthly on the 18th: the crawl for month M lands in BigQuery around -# mid-month M+1, so the 18th reliably picks up the newest complete crawl. -# The fetch script itself also falls back to the newest table it can find. +# mid-month M+1, so the 18th reliably picks up the newest available crawl. +# The fetch script itself also falls back to the newest non-empty partition it can find. # # One-time setup is documented in scripts/adoption/README.md (GCP project + # Workload Identity Federation; no long-lived secrets). @@ -75,7 +75,7 @@ jobs: git push -f -u origin "$branch" body_file=$(mktemp) trap 'rm -f "$body_file"' EXIT - printf 'Monthly refresh of the HTTP Archive adoption numbers rendered on the spec pages (`src/data/adoption.json`, crawl `%s`).\n\nGenerated by `npm run adoption` — a single BigQuery aggregation over the newest `httparchive.pages` crawl table. Merge to deploy.\n' "$crawl" > "$body_file" + printf 'Monthly refresh of the HTTP Archive adoption numbers rendered on the spec pages (`src/data/adoption.json`, crawl `%s`).\n\nGenerated by `npm run adoption` — a single BigQuery aggregation over the newest `httparchive.crawl.pages` crawl partition. Merge to deploy.\n' "$crawl" > "$body_file" if gh pr view "$branch" --json number >/dev/null 2>&1; then gh pr edit "$branch" --body-file "$body_file" else diff --git a/scripts/adoption/README.md b/scripts/adoption/README.md index 7d326d4c..086dc995 100644 --- a/scripts/adoption/README.md +++ b/scripts/adoption/README.md @@ -7,8 +7,9 @@ the spec pages render them automatically. ## One-time setup (Google Cloud) The HTTP Archive dataset is public, but BigQuery bills the _querying_ project, -so you need a small GCP project of your own. One aggregation query per month -fits comfortably inside the 1 TiB/month free tier. +so you need a GCP project of your own. Every refresh first validates the query +and estimates bytes processed with a free dry run. Executed aggregations have a +default 100 GiB billing cap; the free tier is shared with other project usage. 1. Create a GCP project and enable the BigQuery API. 2. Create a service account (no keys needed) with the **BigQuery Job User** @@ -32,12 +33,33 @@ fits comfortably inside the 1 TiB/month free tier. refreshed `src/data/adoption.json`. - Manually: `npm run adoption` (needs `GCP_PROJECT_ID` set and Application Default Credentials, e.g. via `gcloud auth application-default login`). +- Validate only: `GCP_PROJECT_ID=your-project ADOPTION_DRY_RUN=1 npm run adoption`. + This performs a BigQuery dry run and prints the estimated bytes. It never runs + the aggregation or writes `src/data/adoption.json`. +- Select a month explicitly with `ADOPTION_CRAWL=2026-09`; otherwise the script + browses recent partitions with the free table preview API and selects the + newest non-empty one. Availability does not establish crawl completeness. +- Print SQL offline: `ADOPTION_PRINT_QUERY=1 npm run adoption`. No project or + credentials are needed; `ADOPTION_CRAWL` defaults to the current UTC month. +- Override the execution cap with `ADOPTION_MAX_BYTES_BILLED` (a positive number + of bytes; default `107374182400`). - The workflow can also be triggered by hand via workflow_dispatch. +The query reads `httparchive.crawl.pages`, restricted to one date, desktop +clients, and root pages. It counts distinct `root_page` values, retaining the +scheme and port instead of collapsing different origins to a hostname. + +Run regression checks with `node --test scripts/adoption/fetch-adoption.test.mjs`. + ## Adding a topic Add an entry to `metrics.json`: the spec page `slug` plus one or more -conditions against the `custom_metrics` JSON column. Find the exact key and -field shapes in the [custom-metrics repo](https://github.com/HTTPArchive/custom-metrics/tree/main/dist). -The script warns when a configured top-level key is missing from the crawl, so -a typo surfaces on the first run instead of silently reading 0. +conditions against a JSON field inside the `custom_metrics` STRUCT. Each +condition supplies `field` (for example, `well_known`, `robots_txt`, or `other`) +and a `jsonPath` relative to that field. llms.txt results live at +`custom_metrics.other.llms_txt_validation`. Verify names against the live table +schema and preview data; raw metric names differ from the stored field names. +Find metric behaviour in the [custom-metrics repo](https://github.com/HTTPArchive/custom-metrics/tree/main/dist). +A missing STRUCT field fails validation, while missing JSON properties still +produce zero matches. The pending upstream metrics remain unmeasured until a +crawl includes them. diff --git a/scripts/adoption/fetch-adoption.mjs b/scripts/adoption/fetch-adoption.mjs index 395ca2a9..b61747cf 100644 --- a/scripts/adoption/fetch-adoption.mjs +++ b/scripts/adoption/fetch-adoption.mjs @@ -1,130 +1,96 @@ #!/usr/bin/env node /** - * Fetches HTTP Archive custom-metric adoption numbers for spec topics and - * writes them to src/data/adoption.json, which the spec pages render. - * - * Reads its metric definitions from ./metrics.json. Each metric becomes one - * COUNTIF over the latest `httparchive.pages.YYYY_MM_01_desktop` table, so a - * full refresh is a single BigQuery aggregation query. - * - * Auth: Application Default Credentials. Locally that means - * `gcloud auth application-default login`; in CI the adoption workflow - * authenticates via Workload Identity Federation. Query billing lands on - * GCP_PROJECT_ID (the httparchive dataset itself is public). - * - * Usage: GCP_PROJECT_ID=your-project node scripts/adoption/fetch-adoption.mjs + * Aggregate desktop root-page metrics from httparchive.crawl.pages. + * GCP_PROJECT_ID selects the querying project; ADC supplies credentials. + * ADOPTION_DRY_RUN=1 validates and estimates without executing or writing data. */ import { readFileSync, writeFileSync } from "node:fs"; import { dirname, resolve } from "node:path"; -import { fileURLToPath } from "node:url"; +import { fileURLToPath, pathToFileURL } from "node:url"; import { BigQuery } from "@google-cloud/bigquery"; const here = dirname(fileURLToPath(import.meta.url)); -const projectId = process.env.GCP_PROJECT_ID; -if (!projectId) { - console.error( - "error: GCP_PROJECT_ID is required (project billed for the query)", - ); - process.exit(1); -} +const defaultMaximumBytesBilled = String(100 * 1024 ** 3); -const config = JSON.parse(readFileSync(resolve(here, "metrics.json"), "utf8")); -const bigquery = new BigQuery({ projectId }); +function validateCrawl(crawl) { + if (!/^\d{4}-(0[1-9]|1[0-2])$/.test(crawl)) { + throw new Error("ADOPTION_CRAWL must be YYYY-MM"); + } + return crawl; +} -/** Newest httparchive.pages crawl table that actually exists (data lands mid-month). */ -async function latestCrawlTable() { - const now = new Date(); +/** Browse one row per partition via the free tabledata.list API, not a query. */ +export async function latestCrawl(bigquery, now = new Date()) { for (let back = 0; back <= 6; back++) { - const d = new Date( + const date = new Date( Date.UTC(now.getUTCFullYear(), now.getUTCMonth() - back, 1), ); - const suffix = `${d.getUTCFullYear()}_${String(d.getUTCMonth() + 1).padStart(2, "0")}_01_desktop`; + const crawl = date.toISOString().slice(0, 7); try { - await bigquery.query({ - query: `SELECT 1 FROM \`httparchive.pages.${suffix}\` LIMIT 0`, - dryRun: false, - }); - return suffix; + const [rows] = await bigquery + .dataset("crawl", { projectId: "httparchive" }) + .table(`pages$${crawl.replace("-", "")}01`) + .getRows({ maxResults: 1, selectedFields: "date" }); + if (rows.length) return crawl; } catch (error) { if (error?.code !== 404) throw error; } } - throw new Error( - "no httparchive.pages crawl table found in the last 6 months", - ); + throw new Error("no HTTP Archive crawl found in the last 6 months"); +} + +function sqlString(value) { + return `'${String(value).replaceAll("\\", "\\\\").replaceAll("'", "\\'")}'`; } function conditionSql(condition) { + if (!/^[a-z_]+$/.test(condition.field)) { + throw new Error(`invalid custom_metrics field: ${condition.field}`); + } + const input = `custom_metrics.${condition.field}, ${sqlString(condition.jsonPath)}`; if (condition.op === "eq") { - return `JSON_VALUE(custom_metrics, '${condition.jsonPath}') = '${condition.value}'`; + return `JSON_VALUE(${input}) = ${sqlString(condition.value)}`; } if (condition.op === "exists") { - return `JSON_QUERY(custom_metrics, '${condition.jsonPath}') IS NOT NULL`; + // JSON null is a value for native JSON columns; it is not evidence of a rule. + return `COALESCE(JSON_TYPE(JSON_QUERY(${input})) != 'null', FALSE)`; } throw new Error(`unknown condition op: ${condition.op}`); } -function metricSql(metric, index) { - const parts = metric.conditions.map(conditionSql); - const combined = - metric.combine === "all" ? parts.join(" AND ") : parts.join(" OR "); - return `COUNTIF(${combined}) AS pages_m${index}`; -} - -/** Warn when a configured top-level custom-metric key is absent from the crawl. */ -async function checkKeys(table, keys) { - const [rows] = await bigquery.query({ - query: `SELECT custom_metrics FROM \`httparchive.pages.${table}\` LIMIT 1`, - }); - const parsed = JSON.parse(rows[0]?.custom_metrics ?? "{}"); - const missing = [...new Set(keys)].filter((k) => !(k in parsed)); - for (const key of missing) { - console.warn( - `warning: custom metric key "${key}" not present in crawl ${table}; ` + - `those adoption numbers will read 0. Available keys: ${Object.keys(parsed).slice(0, 12).join(", ")}…`, - ); - } -} - -/** Origins-based COUNTIF needs the host per hit; do it with a second pass over hosts. */ -function originsSql(metric, index) { - const parts = metric.conditions.map(conditionSql); - const combined = - metric.combine === "all" ? parts.join(" AND ") : parts.join(" OR "); - return `COUNT(DISTINCT IF(${combined}, NET.HOST(url), NULL)) AS origins_m${index}`; -} - -function buildQuery(table) { +export function buildQuery(config, crawl) { + validateCrawl(crawl); const selects = config.metrics - .flatMap((metric, i) => [metricSql(metric, i), originsSql(metric, i)]) + .flatMap((metric, i) => { + if (!metric.conditions.length) + throw new Error(`no conditions: ${metric.slug}`); + const combined = metric.conditions + .map(conditionSql) + .join(metric.combine === "all" ? " AND " : " OR "); + return [ + `COUNTIF(${combined}) AS pages_m${i}`, + `COUNT(DISTINCT IF(${combined}, root_page, NULL)) AS origins_m${i}`, + ]; + }) .join(",\n "); return `SELECT COUNT(*) AS pages, - COUNT(DISTINCT NET.HOST(url)) AS origins, + COUNT(DISTINCT root_page) AS origins, ${selects} - FROM \`httparchive.pages.${table}\``; + FROM \`httparchive.crawl.pages\` + WHERE date = DATE '${crawl}-01' + AND client = 'desktop' + AND is_root_page`; } -async function main() { - // Offline mode: print the generated SQL without touching BigQuery. - if (process.env.ADOPTION_PRINT_QUERY) { - console.log(buildQuery("2026_09_01_desktop")); - return; +export function buildReport(config, crawl, row) { + const pagesTotal = Number(row?.pages); + const originsTotal = Number(row?.origins); + if (!(pagesTotal > 0) || !(originsTotal > 0)) { + throw new Error( + `crawl ${crawl} has no desktop root pages; keeping existing data`, + ); } - - const table = await latestCrawlTable(); - const crawl = table.slice(0, 7).replace("_", "-"); - console.log(`using crawl table httparchive.pages.${table}`); - - const topKeys = config.metrics.flatMap((m) => - m.conditions.map((c) => c.jsonPath.split(".")[1]), - ); - await checkKeys(table, topKeys); - - const query = buildQuery(table); - const [rows] = await bigquery.query({ query }); - const row = rows[0]; - const metrics = {}; config.metrics.forEach((metric, i) => { const pages = Number(row[`pages_m${i}`]); @@ -132,24 +98,75 @@ async function main() { metrics[metric.slug] = { label: metric.label, pages, - pagesPct: roundPct(pages / Number(row.pages)), + pagesPct: roundPct(pages / pagesTotal), origins, - originsPct: roundPct(origins / Number(row.origins)), + originsPct: roundPct(origins / originsTotal), }; }); - - const out = { + return { crawl, generatedAt: new Date().toISOString(), - source: "HTTP Archive monthly crawl (desktop), custom metrics", - pages: Number(row.pages), - origins: Number(row.origins), + source: "HTTP Archive monthly crawl (desktop root pages), custom metrics", + pages: pagesTotal, + origins: originsTotal, metrics, }; +} + +export async function fetchAdoption(bigquery, config, options = {}) { + const crawl = validateCrawl(options.crawl || (await latestCrawl(bigquery))); + const maximumBytesBilled = + options.maximumBytesBilled || defaultMaximumBytesBilled; + if (!/^[1-9]\d*$/.test(maximumBytesBilled)) { + throw new Error("ADOPTION_MAX_BYTES_BILLED must be a positive integer"); + } + const queryOptions = { + query: buildQuery(config, crawl), + location: "US", + useLegacySql: false, + maximumBytesBilled, + }; + const [job] = await bigquery.createQueryJob({ + ...queryOptions, + dryRun: true, + }); + console.log( + `crawl ${crawl}: dry run passed; estimated bytes processed: ${job.metadata.statistics?.totalBytesProcessed ?? "unknown"}`, + ); + if (options.dryRun) return null; + + const [rows] = await bigquery.query(queryOptions); + return buildReport(config, crawl, rows[0]); +} + +async function main() { + const config = JSON.parse( + readFileSync(resolve(here, "metrics.json"), "utf8"), + ); + if (process.env.ADOPTION_PRINT_QUERY === "1") { + console.log( + buildQuery( + config, + process.env.ADOPTION_CRAWL || new Date().toISOString().slice(0, 7), + ), + ); + return; + } + const projectId = process.env.GCP_PROJECT_ID; + if (!projectId) + throw new Error( + "GCP_PROJECT_ID is required (project billed for the query)", + ); + const report = await fetchAdoption(new BigQuery({ projectId }), config, { + crawl: process.env.ADOPTION_CRAWL, + dryRun: process.env.ADOPTION_DRY_RUN === "1", + maximumBytesBilled: process.env.ADOPTION_MAX_BYTES_BILLED, + }); + if (!report) return; const dest = resolve(here, "../../src/data/adoption.json"); - writeFileSync(dest, JSON.stringify(out, null, 2) + "\n"); + writeFileSync(dest, JSON.stringify(report, null, 2) + "\n"); console.log( - `wrote ${dest} (${config.metrics.length} metrics, crawl ${crawl})`, + `wrote ${dest} (${config.metrics.length} metrics, crawl ${report.crawl})`, ); } @@ -157,7 +174,12 @@ function roundPct(ratio) { return Math.round(ratio * 100 * 1000) / 1000; } -main().catch((error) => { - console.error(`error: ${error.message}`); - process.exit(1); -}); +if ( + process.argv[1] && + import.meta.url === pathToFileURL(resolve(process.argv[1])).href +) { + main().catch((error) => { + console.error(`error: ${error.message}`); + process.exitCode = 1; + }); +} diff --git a/scripts/adoption/fetch-adoption.test.mjs b/scripts/adoption/fetch-adoption.test.mjs new file mode 100644 index 00000000..d85fffb5 --- /dev/null +++ b/scripts/adoption/fetch-adoption.test.mjs @@ -0,0 +1,133 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { execFileSync } from "node:child_process"; +import test from "node:test"; +import { + buildQuery, + buildReport, + fetchAdoption, + latestCrawl, +} from "./fetch-adoption.mjs"; + +const config = JSON.parse( + readFileSync(new URL("./metrics.json", import.meta.url), "utf8"), +); + +test("query uses current JSON fields and keeps HTTP and HTTPS origins distinct", () => { + const query = buildQuery(config, "2026-09"); + assert.match(query, /FROM `httparchive\.crawl\.pages`/); + assert.match(query, /date = DATE '2026-09-01'/); + assert.match(query, /client = 'desktop'\s+AND is_root_page/); + assert.match( + query, + /JSON_VALUE\(custom_metrics\.well_known, '\$\."\/\.well-known\/gpc.json"\.found'\)/, + ); + assert.match( + query, + /JSON_VALUE\(custom_metrics\.other, '\$\.llms_txt_validation\.valid'\)/, + ); + assert.match(query, /COUNT\(DISTINCT root_page\)/); + assert.doesNotMatch(query, /NET\.HOST|\$\._|JSON_VALUE\(custom_metrics,/); +}); + +test("native JSON null is not counted as an existing crawler rule", () => { + assert.match( + buildQuery(config, "2026-09"), + /COALESCE\(JSON_TYPE\(JSON_QUERY\(custom_metrics\.robots_txt, .*?\)\) != 'null', FALSE\)/, + ); +}); + +test("dry run validates but never executes the aggregation or returns publishable data", async () => { + let calls = 0; + const bigquery = { + async createQueryJob(options) { + calls++; + assert.equal(options.dryRun, true); + assert.equal(options.useLegacySql, false); + assert.equal(options.maximumBytesBilled, "107374182400"); + return [{ metadata: { statistics: { totalBytesProcessed: "123" } } }]; + }, + async query() { + assert.fail("A dry run must not execute the query"); + }, + }; + assert.equal( + await fetchAdoption(bigquery, config, { crawl: "2026-09", dryRun: true }), + null, + ); + assert.equal(calls, 1); +}); + +test("crawl discovery skips absent and empty partitions using table previews", async () => { + const partitions = []; + const bigquery = { + dataset() { + return { + table(id) { + partitions.push(id); + return { + async getRows(options) { + assert.deepEqual(options, { + maxResults: 1, + selectedFields: "date", + }); + if (partitions.length === 1) + throw Object.assign(new Error("missing"), { code: 404 }); + return [partitions.length === 2 ? [] : [{ date: "2026-07-01" }]]; + }, + }; + }, + }; + }, + }; + assert.equal( + await latestCrawl(bigquery, new Date("2026-09-28T00:00:00Z")), + "2026-07", + ); + assert.deepEqual(partitions, [ + "pages$20260901", + "pages$20260801", + "pages$20260701", + ]); +}); + +test("crawl discovery does not hide permission failures", async () => { + const denied = Object.assign(new Error("denied"), { code: 403 }); + const bigquery = { + dataset() { + return { + table() { + return { + async getRows() { + throw denied; + }, + }; + }, + }; + }, + }; + await assert.rejects(latestCrawl(bigquery), (error) => error === denied); +}); + +test("empty results cannot replace published adoption data", () => { + assert.throws( + () => buildReport(config, "2026-09", { pages: 0, origins: 0 }), + /keeping existing data/, + ); +}); + +test("offline SQL printing needs neither project configuration nor credentials", () => { + const env = { + ...process.env, + ADOPTION_PRINT_QUERY: "1", + ADOPTION_CRAWL: "2026-09", + }; + delete env.GCP_PROJECT_ID; + const sql = execFileSync( + process.execPath, + [new URL("./fetch-adoption.mjs", import.meta.url).pathname], + { env, encoding: "utf8" }, + ); + assert.match(sql, /httparchive\.crawl\.pages/); + assert.throws(() => buildQuery(config, "2026-13"), /YYYY-MM/); +}); diff --git a/scripts/adoption/metrics.json b/scripts/adoption/metrics.json index 7290fe4e..b0a1bb88 100644 --- a/scripts/adoption/metrics.json +++ b/scripts/adoption/metrics.json @@ -1,12 +1,13 @@ { - "$comment": "Maps spec page slugs to HTTP Archive custom-metric signals. Each metric lists one or more JSONPath conditions against the `custom_metrics` column of `httparchive.pages.YYYY_MM_01_desktop`. The fetch script builds one COUNTIF per metric; `combine: any` counts a page when at least one condition holds.", + "$comment": "Maps spec slugs to HTTP Archive custom_metrics JSON fields in httparchive.crawl.pages. Each condition has a field within that STRUCT and a JSONPath relative to that field. combine: any counts a page when at least one condition holds.", "metrics": [ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/gpc.json\".found", + "jsonPath": "$.\"/.well-known/gpc.json\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/gpc.json", @@ -15,9 +16,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/security.txt\".found", + "jsonPath": "$.\"/.well-known/security.txt\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/security.txt", @@ -26,9 +28,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/assetlinks.json\".found", + "jsonPath": "$.\"/.well-known/assetlinks.json\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/assetlinks.json", @@ -37,9 +40,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/apple-app-site-association\".found", + "jsonPath": "$.\"/.well-known/apple-app-site-association\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/apple-app-site-association", @@ -48,9 +52,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/change-password\".found", + "jsonPath": "$.\"/.well-known/change-password\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/change-password", @@ -59,9 +64,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/webauthn\".found", + "jsonPath": "$.\"/.well-known/webauthn\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/webauthn", @@ -70,9 +76,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/ai-catalog.json\".found", + "jsonPath": "$.\"/.well-known/ai-catalog.json\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/ai-catalog.json (agent catalog)", @@ -81,9 +88,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/ard.json\".found", + "jsonPath": "$.\"/.well-known/ard.json\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/ard.json", @@ -93,9 +101,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/nodeinfo\".found", + "jsonPath": "$.\"/.well-known/nodeinfo\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/nodeinfo", @@ -105,9 +114,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/webfinger\".found", + "jsonPath": "$.\"/.well-known/webfinger\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/webfinger", @@ -117,9 +127,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/oauth-authorization-server\".found", + "jsonPath": "$.\"/.well-known/oauth-authorization-server\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/oauth-authorization-server", @@ -129,9 +140,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/oauth-protected-resource\".found", + "jsonPath": "$.\"/.well-known/oauth-protected-resource\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/oauth-protected-resource", @@ -141,9 +153,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/openid-configuration\".found", + "jsonPath": "$.\"/.well-known/openid-configuration\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/openid-configuration", @@ -153,9 +166,10 @@ { "conditions": [ { - "jsonPath": "$._well-known.\"/.well-known/traffic-advice\".found", + "jsonPath": "$.\"/.well-known/traffic-advice\".found", "op": "eq", - "value": "true" + "value": "true", + "field": "well_known" } ], "label": "/.well-known/traffic-advice", @@ -164,14 +178,24 @@ }, { "conditions": [ - { "jsonPath": "$._llms-txt-valid.valid", "op": "eq", "value": "true" } + { + "jsonPath": "$.llms_txt_validation.valid", + "op": "eq", + "value": "true", + "field": "other" + } ], "label": "/llms.txt (valid)", "slug": "llms-txt" }, { "conditions": [ - { "jsonPath": "$._robots_txt.status", "op": "eq", "value": "200" } + { + "jsonPath": "$.status", + "op": "eq", + "value": "200", + "field": "robots_txt" + } ], "label": "/robots.txt (HTTP 200)", "slug": "robots-txt" @@ -180,32 +204,39 @@ "combine": "any", "conditions": [ { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"gptbot\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"gptbot\"", + "op": "exists", + "field": "robots_txt" }, { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"claudebot\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"claudebot\"", + "op": "exists", + "field": "robots_txt" }, { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"ccbot\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"ccbot\"", + "op": "exists", + "field": "robots_txt" }, { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"anthropic-ai\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"anthropic-ai\"", + "op": "exists", + "field": "robots_txt" }, { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"google-extended\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"google-extended\"", + "op": "exists", + "field": "robots_txt" }, { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"perplexitybot\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"perplexitybot\"", + "op": "exists", + "field": "robots_txt" }, { - "jsonPath": "$._robots_txt.record_counts.by_useragent.\"bytespider\"", - "op": "exists" + "jsonPath": "$.record_counts.by_useragent.\"bytespider\"", + "op": "exists", + "field": "robots_txt" } ], "label": "robots.txt rules naming AI crawlers", From e580234af8f441a8724cc3528029c57145e00f2b Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Tue, 29 Sep 2026 16:23:21 +0200 Subject: [PATCH 4/8] Limit adoption query to top million and exclude llms.txt --- scripts/adoption/README.md | 23 +++++++++++++++-------- scripts/adoption/fetch-adoption.mjs | 8 +++++--- scripts/adoption/fetch-adoption.test.mjs | 7 ++++--- scripts/adoption/metrics.json | 12 ------------ src/data/adoption.json | 2 +- src/layouts/SpecLayout.astro | 3 ++- 6 files changed, 27 insertions(+), 28 deletions(-) diff --git a/scripts/adoption/README.md b/scripts/adoption/README.md index 086dc995..9266cf61 100644 --- a/scripts/adoption/README.md +++ b/scripts/adoption/README.md @@ -1,8 +1,10 @@ # HTTP Archive adoption data Monthly adoption percentages for spec topics, pulled from the HTTP Archive's -custom metrics via BigQuery. The numbers land in `src/data/adoption.json` and -the spec pages render them automatically. +custom metrics via BigQuery, limited to desktop root pages ranked in the top +million (`rank <= 1000000`). The percentages describe this sample, not the entire +crawl. The numbers land in `src/data/adoption.json`, and the spec pages render +them automatically. ## One-time setup (Google Cloud) @@ -46,8 +48,12 @@ default 100 GiB billing cap; the free tier is shared with other project usage. - The workflow can also be triggered by hand via workflow_dispatch. The query reads `httparchive.crawl.pages`, restricted to one date, desktop -clients, and root pages. It counts distinct `root_page` values, retaining the -scheme and port instead of collapsing different origins to a hostname. +clients, root pages, and `rank <= 1000000`. It counts distinct `root_page` values, +retaining the scheme and port instead of collapsing different origins to a hostname. + +`llms.txt` is excluded because its metric requires reading the large +`custom_metrics.other` JSON column. The remaining 16 topics use only +`custom_metrics.well_known` and `custom_metrics.robots_txt`. Run regression checks with `node --test scripts/adoption/fetch-adoption.test.mjs`. @@ -55,10 +61,11 @@ Run regression checks with `node --test scripts/adoption/fetch-adoption.test.mjs Add an entry to `metrics.json`: the spec page `slug` plus one or more conditions against a JSON field inside the `custom_metrics` STRUCT. Each -condition supplies `field` (for example, `well_known`, `robots_txt`, or `other`) -and a `jsonPath` relative to that field. llms.txt results live at -`custom_metrics.other.llms_txt_validation`. Verify names against the live table -schema and preview data; raw metric names differ from the stored field names. +condition supplies `field` (`well_known` or `robots_txt`) and a `jsonPath` +relative to that field. Adding another column can substantially increase bytes +processed; check the free dry-run estimate before extending the query. Verify +names against the live table schema and preview data; raw metric names differ +from the stored field names. Find metric behaviour in the [custom-metrics repo](https://github.com/HTTPArchive/custom-metrics/tree/main/dist). A missing STRUCT field fails validation, while missing JSON properties still produce zero matches. The pending upstream metrics remain unmeasured until a diff --git a/scripts/adoption/fetch-adoption.mjs b/scripts/adoption/fetch-adoption.mjs index b61747cf..ecb4c2c8 100644 --- a/scripts/adoption/fetch-adoption.mjs +++ b/scripts/adoption/fetch-adoption.mjs @@ -1,6 +1,6 @@ #!/usr/bin/env node /** - * Aggregate desktop root-page metrics from httparchive.crawl.pages. + * Aggregate top-million desktop root-page metrics from httparchive.crawl.pages. * GCP_PROJECT_ID selects the querying project; ADC supplies credentials. * ADOPTION_DRY_RUN=1 validates and estimates without executing or writing data. */ @@ -80,7 +80,8 @@ export function buildQuery(config, crawl) { FROM \`httparchive.crawl.pages\` WHERE date = DATE '${crawl}-01' AND client = 'desktop' - AND is_root_page`; + AND is_root_page + AND rank <= 1000000`; } export function buildReport(config, crawl, row) { @@ -106,7 +107,8 @@ export function buildReport(config, crawl, row) { return { crawl, generatedAt: new Date().toISOString(), - source: "HTTP Archive monthly crawl (desktop root pages), custom metrics", + source: + "HTTP Archive monthly crawl (desktop root pages, rank <= 1000000), custom metrics", pages: pagesTotal, origins: originsTotal, metrics, diff --git a/scripts/adoption/fetch-adoption.test.mjs b/scripts/adoption/fetch-adoption.test.mjs index d85fffb5..8565c191 100644 --- a/scripts/adoption/fetch-adoption.test.mjs +++ b/scripts/adoption/fetch-adoption.test.mjs @@ -17,15 +17,16 @@ test("query uses current JSON fields and keeps HTTP and HTTPS origins distinct", const query = buildQuery(config, "2026-09"); assert.match(query, /FROM `httparchive\.crawl\.pages`/); assert.match(query, /date = DATE '2026-09-01'/); - assert.match(query, /client = 'desktop'\s+AND is_root_page/); assert.match( query, - /JSON_VALUE\(custom_metrics\.well_known, '\$\."\/\.well-known\/gpc.json"\.found'\)/, + /client = 'desktop'\s+AND is_root_page\s+AND rank <= 1000000/, ); assert.match( query, - /JSON_VALUE\(custom_metrics\.other, '\$\.llms_txt_validation\.valid'\)/, + /JSON_VALUE\(custom_metrics\.well_known, '\$\."\/\.well-known\/gpc.json"\.found'\)/, ); + assert.doesNotMatch(query, /custom_metrics\.other|llms_txt_validation/); + assert.ok(!config.metrics.some((metric) => metric.slug === "llms-txt")); assert.match(query, /COUNT\(DISTINCT root_page\)/); assert.doesNotMatch(query, /NET\.HOST|\$\._|JSON_VALUE\(custom_metrics,/); }); diff --git a/scripts/adoption/metrics.json b/scripts/adoption/metrics.json index b0a1bb88..4b52e87e 100644 --- a/scripts/adoption/metrics.json +++ b/scripts/adoption/metrics.json @@ -176,18 +176,6 @@ "note": "Needs the well-known metric extension to land upstream first.", "slug": "traffic-advice" }, - { - "conditions": [ - { - "jsonPath": "$.llms_txt_validation.valid", - "op": "eq", - "value": "true", - "field": "other" - } - ], - "label": "/llms.txt (valid)", - "slug": "llms-txt" - }, { "conditions": [ { diff --git a/src/data/adoption.json b/src/data/adoption.json index d77247c4..f7611948 100644 --- a/src/data/adoption.json +++ b/src/data/adoption.json @@ -4,5 +4,5 @@ "metrics": {}, "origins": 0, "pages": 0, - "source": "HTTP Archive monthly crawl (desktop), custom metrics" + "source": "HTTP Archive monthly crawl (desktop root pages, rank <= 1000000), custom metrics" } diff --git a/src/layouts/SpecLayout.astro b/src/layouts/SpecLayout.astro index 09791cce..475610dc 100644 --- a/src/layouts/SpecLayout.astro +++ b/src/layouts/SpecLayout.astro @@ -161,7 +161,8 @@ const breadcrumbJsonLd = {

Adoption:{" "} {adoptionLabel} of - origins in the HTTP Archive {crawlLabel} crawl + origins ranked in the top million in the HTTP Archive {crawlLabel}{" "} + desktop crawl

)} {appliesTo.length > 0 && appliesTo[0] !== "all" && ( From 2ade7d9d0f1111472f1d8837bb89cd0cadffe3b4 Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Tue, 29 Sep 2026 16:36:18 +0200 Subject: [PATCH 5/8] Use parsed adoption signals and correct catalogue mapping --- .github/workflows/ci.yml | 3 + package.json | 3 +- scripts/adoption/README.md | 56 +- scripts/adoption/fetch-adoption.mjs | 17 +- scripts/adoption/fetch-adoption.test.mjs | 56 ++ scripts/adoption/fixtures.json | 912 +++++++++++++++++++++++ scripts/adoption/metrics.json | 287 +++---- scripts/adoption/query-fixtures.test.mjs | 99 +++ 8 files changed, 1291 insertions(+), 142 deletions(-) create mode 100644 scripts/adoption/fixtures.json create mode 100644 scripts/adoption/query-fixtures.test.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c7614dd6..0523532f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -64,6 +64,9 @@ jobs: - name: Agent Skill drift check run: npm run check:skill + - name: Test adoption query + run: npm run test:adoption + - name: Type-check (Astro) run: npx astro check diff --git a/package.json b/package.json index 1efab0ea..dd634b8c 100644 --- a/package.json +++ b/package.json @@ -31,7 +31,8 @@ "test:websub": "node --experimental-strip-types scripts/test-websub.mjs", "sign:skill": "node scripts/check-skill.mjs", "check:skill": "node scripts/check-skill.mjs --check", - "adoption": "node scripts/adoption/fetch-adoption.mjs" + "adoption": "node scripts/adoption/fetch-adoption.mjs", + "test:adoption": "node --test scripts/adoption/*.test.mjs" }, "dependencies": { "@astrojs/mdx": "^8.0.2", diff --git a/scripts/adoption/README.md b/scripts/adoption/README.md index 9266cf61..4e19750d 100644 --- a/scripts/adoption/README.md +++ b/scripts/adoption/README.md @@ -52,21 +52,65 @@ clients, root pages, and `rank <= 1000000`. It counts distinct `root_page` value retaining the scheme and port instead of collapsing different origins to a hostname. `llms.txt` is excluded because its metric requires reading the large -`custom_metrics.other` JSON column. The remaining 16 topics use only +`custom_metrics.other` JSON column. The nine measured topics use only `custom_metrics.well_known` and `custom_metrics.robots_txt`. -Run regression checks with `node --test scripts/adoption/fetch-adoption.test.mjs`. +Run regression checks with `npm run test:adoption` (also run by CI). To verify +SQL behaviour in BigQuery with the recorded collector fixtures, run +`ADOPTION_TEST_PROJECT=your-project npm run test:adoption`. This optional test +reads only inline parameters, asserts a zero-byte dry run before execution, and +never queries HTTP Archive or writes adoption data. + +## What the counts mean + +These are conservative detection signals, not full conformance checks. A 200 +response alone is insufficient: generic HTML and JSON catch-all pages must not +count as adoption. + +| Topic | Required evidence, in addition to a successful response | +| -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| GPC support resource | The parsed `gpc` declaration is non-null. Both boolean values count as a published declaration; this does not establish that GPC is honoured. | +| security.txt | The collector's `data.valid` field is true: required fields exist and singleton fields are not repeated. Values and expiry are not fully validated. | +| Asset Links | A parsed deep-linking or credential-sharing relation is detected. | +| Apple App Site Association | Parsed app-link or web-credential configuration is detected. | +| Change password | A followed redirect ends in 200, and the collector's deliberately nonexistent URL returns 404. Failed or missing probes are not positive evidence. | +| WebAuthn related origins | A non-empty parsed `origins` array. Individual origins are not validated by the collector. | +| ARD | At least one parsed entry at either `ai-catalog.json` or `ard.json`. An origin with both counts once. Empty manifests cannot be distinguished from parser defaults and are excluded. | +| robots.txt | At least one parsed user-agent or sitemap directive. Empty or comments-only files are excluded. | +| AI crawler rules | A successful robots.txt response with a parsed group naming one of the configured AI crawlers. This detects a named group, not whether crawling is allowed or disallowed. | + +The [well-known collector](https://github.com/HTTPArchive/custom-metrics/blob/65e46f181eded5e9aeb26b53b88681361093eae4/dist/well-known.js) +and [robots.txt collector](https://github.com/HTTPArchive/custom-metrics/blob/1e9b5f6a8cd3576befcb770f46e210d2b0bfb243/dist/robots_txt.js) +define the stored signals. `fixtures.json` records their outputs for mocked +responses, including generic HTML, redirects, JSON catch-alls, valid documents, +and failed probes. The live SQL test also checks rank, client, date, root-page +filtering, and distinct-origin counts. + +The [RFC 9727 API Catalog](https://www.rfc-editor.org/rfc/rfc9727.html) uses +`/.well-known/api-catalog`; it is not the AI Catalog. No corresponding collector +signal exists, so the API Catalog topic is omitted. NodeInfo, WebFinger, OAuth +authorisation-server metadata, OAuth protected-resource metadata, OpenID +Configuration, and Traffic Advice are also omitted while their collector +extension is pending. They must not be reported as zero adoption. Add them only +after reviewing the collected fields and validating their detection rules. + +Older crawls may lack newer parser fields (including ARD entry counts); such +responses cannot establish adoption under these rules. The denominator remains +all sampled origins, so percentages describe detected signals in that sample +and may undercount actual implementations. ## Adding a topic Add an entry to `metrics.json`: the spec page `slug` plus one or more conditions against a JSON field inside the `custom_metrics` STRUCT. Each condition supplies `field` (`well_known` or `robots_txt`) and a `jsonPath` -relative to that field. Adding another column can substantially increase bytes +relative to that field. Conditions combine with `all` by default; nested groups +can use `any` for alternative signals. Supported operations are `eq`, `exists`, +`positive`, and `nonempty-array`. Adding another column can substantially increase bytes processed; check the free dry-run estimate before extending the query. Verify names against the live table schema and preview data; raw metric names differ from the stored field names. Find metric behaviour in the [custom-metrics repo](https://github.com/HTTPArchive/custom-metrics/tree/main/dist). -A missing STRUCT field fails validation, while missing JSON properties still -produce zero matches. The pending upstream metrics remain unmeasured until a -crawl includes them. +A missing STRUCT field fails validation, while missing JSON properties produce +no match. Missing collection is not evidence of zero real-world adoption; do +not add placeholder metrics for fields the collector does not yet provide. diff --git a/scripts/adoption/fetch-adoption.mjs b/scripts/adoption/fetch-adoption.mjs index ecb4c2c8..0b1a726c 100644 --- a/scripts/adoption/fetch-adoption.mjs +++ b/scripts/adoption/fetch-adoption.mjs @@ -44,6 +44,13 @@ function sqlString(value) { } function conditionSql(condition) { + if (condition.conditions) { + if (!condition.conditions.length) throw new Error("empty condition group"); + if (condition.combine && !["all", "any"].includes(condition.combine)) { + throw new Error(`invalid condition combination: ${condition.combine}`); + } + return `(${condition.conditions.map(conditionSql).join(condition.combine === "any" ? " OR " : " AND ")})`; + } if (!/^[a-z_]+$/.test(condition.field)) { throw new Error(`invalid custom_metrics field: ${condition.field}`); } @@ -55,6 +62,12 @@ function conditionSql(condition) { // JSON null is a value for native JSON columns; it is not evidence of a rule. return `COALESCE(JSON_TYPE(JSON_QUERY(${input})) != 'null', FALSE)`; } + if (condition.op === "positive") { + return `SAFE_CAST(JSON_VALUE(${input}) AS FLOAT64) > 0`; + } + if (condition.op === "nonempty-array") { + return `ARRAY_LENGTH(JSON_QUERY_ARRAY(${input})) > 0`; + } throw new Error(`unknown condition op: ${condition.op}`); } @@ -64,9 +77,7 @@ export function buildQuery(config, crawl) { .flatMap((metric, i) => { if (!metric.conditions.length) throw new Error(`no conditions: ${metric.slug}`); - const combined = metric.conditions - .map(conditionSql) - .join(metric.combine === "all" ? " AND " : " OR "); + const combined = conditionSql(metric); return [ `COUNTIF(${combined}) AS pages_m${i}`, `COUNT(DISTINCT IF(${combined}, root_page, NULL)) AS origins_m${i}`, diff --git a/scripts/adoption/fetch-adoption.test.mjs b/scripts/adoption/fetch-adoption.test.mjs index 8565c191..fef81f52 100644 --- a/scripts/adoption/fetch-adoption.test.mjs +++ b/scripts/adoption/fetch-adoption.test.mjs @@ -132,3 +132,59 @@ test("offline SQL printing needs neither project configuration nor credentials", assert.match(sql, /httparchive\.crawl\.pages/); assert.throws(() => buildQuery(config, "2026-13"), /YYYY-MM/); }); + +test("API Catalog is not measured using the unrelated AI Catalog", () => { + assert.ok(!config.metrics.some((metric) => metric.slug === "api-catalog")); + const ard = config.metrics.find( + (metric) => metric.slug === "agentic-resource-discovery", + ); + const sql = buildQuery({ metrics: [ard] }, "2026-09"); + assert.match(sql, /ai-catalog\.json/); + assert.match(sql, /ard\.json/); + assert.match(sql, /entries_count/); + assert.match(sql, /\) OR \(/); +}); + +test("uncollected topics cannot be published as zero adoption", () => { + const report = buildReport(config, "2026-09", { + pages: 100, + origins: 100, + ...Object.fromEntries( + config.metrics.flatMap((_, i) => [ + [`pages_m${i}`, 0], + [`origins_m${i}`, 0], + ]), + ), + }); + for (const slug of [ + "api-catalog", + "nodeinfo", + "webfinger", + "oauth-authorization-server", + "oauth-protected-resource", + "openid-configuration", + "traffic-advice", + "llms-txt", + ]) { + assert.ok(!(slug in report.metrics), slug); + } +}); + +test("empty or invalid condition groups fail before querying", () => { + for (const condition of [ + { conditions: [] }, + { + combine: "none", + conditions: [{ field: "well_known", jsonPath: "$.x", op: "exists" }], + }, + ]) { + assert.throws( + () => + buildQuery( + { metrics: [{ slug: "bad", conditions: [condition] }] }, + "2026-09", + ), + /condition/, + ); + } +}); diff --git a/scripts/adoption/fixtures.json b/scripts/adoption/fixtures.json new file mode 100644 index 00000000..d10edc06 --- /dev/null +++ b/scripts/adoption/fixtures.json @@ -0,0 +1,912 @@ +{ + "source": "HTTPArchive/custom-metrics dist/well-known.js at 65e46f181eded5e9aeb26b53b88681361093eae4; dist/robots_txt.js at 1e9b5f6a8cd3576befcb770f46e210d2b0bfb243. Recorded parser outputs from mocked responses; no live sites.", + "cases": [ + { + "name": "html-200", + "well_known": { + "/.well-known/assetlinks.json": { + "error": "Unexpected token '<', \" { + const allPositive = fixtures.cases.find( + (fixture) => fixture.name === "all-positive", + ); + const cases = [...fixtures.cases, allPositive].map((fixture) => ({ + name: fixture.name, + well_known: JSON.stringify(fixture.well_known), + robots_txt: JSON.stringify(fixture.robots_txt), + crawl_date: "2026-09-01", + client: "desktop", + is_root_page: true, + rank: 1000000, + })); + const positiveRow = cases.at(-1); + for (const [name, overrides] of Object.entries({ + "outside-rank": { rank: 1000001 }, + mobile: { client: "mobile" }, + "non-root": { is_root_page: false }, + "other-month": { crawl_date: "2026-08-01" }, + })) { + cases.push({ ...positiveRow, name, ...overrides }); + } + const sql = buildQuery(config, "2026-09") + .replace("SELECT\n", "SELECT root_page AS fixture,\n") + .replace("`httparchive.crawl.pages`", "fixtures"); + const query = `WITH fixtures AS ( + SELECT name AS root_page, DATE(crawl_date) AS date, + client, is_root_page, rank, + STRUCT(PARSE_JSON(well_known) AS well_known, + PARSE_JSON(robots_txt) AS robots_txt) AS custom_metrics + FROM UNNEST(@cases) + ) ${sql} GROUP BY GROUPING SETS ((root_page), ())`; + assert.doesNotMatch(query, /httparchive\.crawl\.pages/); + const bigquery = new BigQuery({ + projectId: process.env.ADOPTION_TEST_PROJECT, + }); + const options = { + query, + params: { cases }, + useLegacySql: false, + location: "US", + maximumBytesBilled: "10485760", + }; + const [dryRun] = await bigquery.createQueryJob({ + ...options, + dryRun: true, + }); + assert.equal(dryRun.metadata.statistics.totalBytesProcessed, "0"); + const [rows] = await bigquery.query(options); + assert.equal(rows.length, fixtures.cases.length + 1); + for (const fixture of fixtures.cases) { + const row = rows.find((row) => row.fixture === fixture.name); + assert.ok(row, fixture.name); + const duplicates = fixture.name === "all-positive" ? 2 : 1; + assert.equal(Number(row.pages), duplicates, fixture.name); + assert.equal(Number(row.origins), 1, fixture.name); + config.metrics.forEach((metric, i) => { + const expected = fixture.expected.includes(metric.slug) ? 1 : 0; + assert.equal( + Number(row[`origins_m${i}`]), + expected, + `${fixture.name}: ${metric.slug}`, + ); + assert.equal( + Number(row[`pages_m${i}`]), + expected * duplicates, + `${fixture.name}: ${metric.slug} pages`, + ); + }); + } + const totals = rows.find((row) => row.fixture === null); + const report = buildReport(config, "2026-09", totals); + assert.equal(report.origins, fixtures.cases.length); + assert.equal(report.pages, fixtures.cases.length + 1); + assert.equal(report.metrics["agentic-resource-discovery"].origins, 4); + assert.equal(report.metrics["agentic-resource-discovery"].originsPct, 20); + assert.ok(!("api-catalog" in report.metrics)); + }, +); From 1c210078613ae2b745a0a0696a0e6d70c0b3eeef Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Wed, 30 Sep 2026 21:07:53 +0200 Subject: [PATCH 6/8] Publish verified September adoption data with 112 GiB cap --- scripts/adoption/README.md | 8 ++- scripts/adoption/fetch-adoption.mjs | 2 +- scripts/adoption/fetch-adoption.test.mjs | 2 +- src/data/adoption.json | 76 ++++++++++++++++++++++-- 4 files changed, 78 insertions(+), 10 deletions(-) diff --git a/scripts/adoption/README.md b/scripts/adoption/README.md index 4e19750d..b1c5f626 100644 --- a/scripts/adoption/README.md +++ b/scripts/adoption/README.md @@ -11,7 +11,11 @@ them automatically. The HTTP Archive dataset is public, but BigQuery bills the _querying_ project, so you need a GCP project of your own. Every refresh first validates the query and estimates bytes processed with a free dry run. Executed aggregations have a -default 100 GiB billing cap; the free tier is shared with other project usage. +default 112 GiB billing cap; the free tier is shared with other project usage. +BigQuery checks an upper-bound estimate for this clustered table before running +it. The September 2026 query was rejected at 100 GiB but succeeded at 112 GiB, +actually processing 1,634,231,969 bytes (1.522 GiB). The cap is a ceiling, not the +expected scan size; future monthly scans may differ. 1. Create a GCP project and enable the BigQuery API. 2. Create a service account (no keys needed) with the **BigQuery Job User** @@ -44,7 +48,7 @@ default 100 GiB billing cap; the free tier is shared with other project usage. - Print SQL offline: `ADOPTION_PRINT_QUERY=1 npm run adoption`. No project or credentials are needed; `ADOPTION_CRAWL` defaults to the current UTC month. - Override the execution cap with `ADOPTION_MAX_BYTES_BILLED` (a positive number - of bytes; default `107374182400`). + of bytes; default `120259084288`). - The workflow can also be triggered by hand via workflow_dispatch. The query reads `httparchive.crawl.pages`, restricted to one date, desktop diff --git a/scripts/adoption/fetch-adoption.mjs b/scripts/adoption/fetch-adoption.mjs index 0b1a726c..665ac385 100644 --- a/scripts/adoption/fetch-adoption.mjs +++ b/scripts/adoption/fetch-adoption.mjs @@ -10,7 +10,7 @@ import { fileURLToPath, pathToFileURL } from "node:url"; import { BigQuery } from "@google-cloud/bigquery"; const here = dirname(fileURLToPath(import.meta.url)); -const defaultMaximumBytesBilled = String(100 * 1024 ** 3); +const defaultMaximumBytesBilled = String(112 * 1024 ** 3); function validateCrawl(crawl) { if (!/^\d{4}-(0[1-9]|1[0-2])$/.test(crawl)) { diff --git a/scripts/adoption/fetch-adoption.test.mjs b/scripts/adoption/fetch-adoption.test.mjs index fef81f52..dd54a104 100644 --- a/scripts/adoption/fetch-adoption.test.mjs +++ b/scripts/adoption/fetch-adoption.test.mjs @@ -45,7 +45,7 @@ test("dry run validates but never executes the aggregation or returns publishabl calls++; assert.equal(options.dryRun, true); assert.equal(options.useLegacySql, false); - assert.equal(options.maximumBytesBilled, "107374182400"); + assert.equal(options.maximumBytesBilled, "120259084288"); return [{ metadata: { statistics: { totalBytesProcessed: "123" } } }]; }, async query() { diff --git a/src/data/adoption.json b/src/data/adoption.json index f7611948..7998d195 100644 --- a/src/data/adoption.json +++ b/src/data/adoption.json @@ -1,8 +1,72 @@ { - "crawl": null, - "generatedAt": null, - "metrics": {}, - "origins": 0, - "pages": 0, - "source": "HTTP Archive monthly crawl (desktop root pages, rank <= 1000000), custom metrics" + "crawl": "2026-09", + "generatedAt": "2026-09-30T19:03:52.787Z", + "source": "HTTP Archive monthly crawl (desktop root pages, rank <= 1000000), custom metrics", + "pages": 668524, + "origins": 668523, + "metrics": { + "gpc-json": { + "label": "gpc.json with a boolean declaration", + "pages": 40017, + "pagesPct": 5.986, + "origins": 40017, + "originsPct": 5.986 + }, + "security-txt": { + "label": "security.txt passing the collector’s field checks", + "pages": 17997, + "pagesPct": 2.692, + "origins": 17997, + "originsPct": 2.692 + }, + "assetlinks-json": { + "label": "/.well-known/assetlinks.json", + "pages": 37595, + "pagesPct": 5.624, + "origins": 37595, + "originsPct": 5.624 + }, + "apple-app-site-association": { + "label": "/.well-known/apple-app-site-association", + "pages": 81422, + "pagesPct": 12.179, + "origins": 81422, + "originsPct": 12.179 + }, + "change-password": { + "label": "change-password redirect with a successful missing-URL check", + "pages": 3910, + "pagesPct": 0.585, + "origins": 3910, + "originsPct": 0.585 + }, + "webauthn": { + "label": "webauthn with a non-empty origins array", + "pages": 345, + "pagesPct": 0.052, + "origins": 345, + "originsPct": 0.052 + }, + "agentic-resource-discovery": { + "label": "AI Catalog or ARD manifest with at least one entry", + "pages": 0, + "pagesPct": 0, + "origins": 0, + "originsPct": 0 + }, + "robots-txt": { + "label": "robots.txt with parsed user-agent or sitemap directives", + "pages": 499464, + "pagesPct": 74.711, + "origins": 499463, + "originsPct": 74.711 + }, + "robots-for-ai-crawlers": { + "label": "robots.txt rules naming AI crawlers", + "pages": 87304, + "pagesPct": 13.059, + "origins": 87303, + "originsPct": 13.059 + } + } } From f757c0e5d0511c5f08bf726a67609ad0a3596992 Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Wed, 30 Sep 2026 21:15:13 +0200 Subject: [PATCH 7/8] Describe adoption sample as websites --- src/layouts/SpecLayout.astro | 6 ++++-- src/pages/spec/[category]/[slug].astro | 2 ++ 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/src/layouts/SpecLayout.astro b/src/layouts/SpecLayout.astro index 475610dc..95d897d3 100644 --- a/src/layouts/SpecLayout.astro +++ b/src/layouts/SpecLayout.astro @@ -28,6 +28,7 @@ interface Props { originsPct: number; } | null; adoptionCrawl?: string | null; + adoptionSampleSize?: number; } const { @@ -42,6 +43,7 @@ const { slug, adoption = null, adoptionCrawl = null, + adoptionSampleSize = 0, } = Astro.props; const adoptionLabel = @@ -161,8 +163,8 @@ const breadcrumbJsonLd = {

Adoption:{" "} {adoptionLabel} of - origins ranked in the top million in the HTTP Archive {crawlLabel}{" "} - desktop crawl + {adoptionSampleSize.toLocaleString("en-GB")} websites in the HTTP + Archive {crawlLabel} desktop sample

)} {appliesTo.length > 0 && appliesTo[0] !== "all" && ( diff --git a/src/pages/spec/[category]/[slug].astro b/src/pages/spec/[category]/[slug].astro index c81ef0ae..0af897b1 100644 --- a/src/pages/spec/[category]/[slug].astro +++ b/src/pages/spec/[category]/[slug].astro @@ -14,6 +14,7 @@ interface AdoptionMetric { interface AdoptionData { crawl: string | null; generatedAt: string | null; + origins: number; metrics: Record; } @@ -64,6 +65,7 @@ const adoption = adoptionData.metrics[slug] ?? null; slug={slug} adoption={adoption} adoptionCrawl={adoptionData.crawl} + adoptionSampleSize={adoptionData.origins} > From a56f45553335cc2d8c14683d3b92bee05b52e224 Mon Sep 17 00:00:00 2001 From: Joost de Valk Date: Wed, 30 Sep 2026 21:16:27 +0200 Subject: [PATCH 8/8] Preserve spacing before adoption sample size --- src/layouts/SpecLayout.astro | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/layouts/SpecLayout.astro b/src/layouts/SpecLayout.astro index 95d897d3..a4191980 100644 --- a/src/layouts/SpecLayout.astro +++ b/src/layouts/SpecLayout.astro @@ -162,7 +162,7 @@ const breadcrumbJsonLd = { {adoptionLabel && crawlLabel && (

Adoption:{" "} - {adoptionLabel} of + {adoptionLabel} of{" "} {adoptionSampleSize.toLocaleString("en-GB")} websites in the HTTP Archive {crawlLabel} desktop sample