diff --git a/.github/workflows/notify-automation-failures.yml b/.github/workflows/notify-automation-failures.yml index 9d1b12a4bb..963a59112d 100644 --- a/.github/workflows/notify-automation-failures.yml +++ b/.github/workflows/notify-automation-failures.yml @@ -19,6 +19,7 @@ on: - Delete Visual Tests Reports - Warm Build Cache - Environment config drift + - Snipsync Coverage types: - completed @@ -27,9 +28,9 @@ permissions: jobs: notify: - # `Check Metrics Against SDKs` and `Environment config drift` also run on - # pull_request; exclude those runs so this only reports invocations that have - # nowhere else to surface, such as the scheduled ones. + # `Check Metrics Against SDKs`, `Environment config drift`, and `Snipsync + # Coverage` also run on pull_request; exclude those runs so this only reports + # invocations that have nowhere else to surface, such as the scheduled ones. if: >- github.event.workflow_run.event != 'pull_request' && contains(fromJSON('["failure", "timed_out", "startup_failure"]'), github.event.workflow_run.conclusion) diff --git a/.github/workflows/snipsync-coverage.yml b/.github/workflows/snipsync-coverage.yml new file mode 100644 index 0000000000..5724d62cf2 --- /dev/null +++ b/.github/workflows/snipsync-coverage.yml @@ -0,0 +1,113 @@ +name: Snipsync Coverage + +# Reports the share of SDK code lines in docs/ that Snipsync pulls from a +# sample repository, as opposed to code written into the page by hand. This is +# informational, not a merge gate: the job never fails on its numbers. +# +# On a pull request it compares the merge result with the base it merges into, +# writes the numbers to the job summary, and keeps one PR comment up to date. +# It only posts that comment when the SDK line counts change, so PRs that don't +# touch code samples stay quiet. On main it writes the numbers to the job +# summary. The numbers depend only on the files in docs/, so the full trend is +# rebuilt from git history with `yarn report:snipsync-coverage --history` +# rather than stored anywhere. + +on: + pull_request: + paths: + - "docs/**" + - "bin/code-blocks.js" + - "bin/report-snipsync-coverage.js" + - ".github/workflows/snipsync-coverage.yml" + push: + branches: [main] + paths: + - "docs/**" + +permissions: + contents: read + pull-requests: write + +concurrency: + group: snipsync-coverage-${{ github.ref }} + cancel-in-progress: true + +jobs: + snipsync-coverage: + name: Report Snipsync coverage + runs-on: ubuntu-latest + steps: + - name: Check out repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # On a pull request HEAD is the merge commit, and HEAD^1 is the base + # it merges into, so the comparison covers exactly this PR's changes. + fetch-depth: 2 + + - name: Set up Node.js + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: "24" + + - name: Report coverage + env: + EVENT_NAME: ${{ github.event_name }} + run: | + if [ "$EVENT_NAME" = "pull_request" ]; then + base="$(git rev-parse HEAD^1)" + node bin/report-snipsync-coverage.js --base "$base" --ref HEAD --markdown > report.md + node bin/report-snipsync-coverage.js --base "$base" --ref HEAD --json > report.json + else + node bin/report-snipsync-coverage.js --markdown > report.md + fi + cat report.md + cat report.md >> "$GITHUB_STEP_SUMMARY" + + - name: Comment on PR + if: github.event_name == 'pull_request' + # A pull request from a fork gets a read-only token and can't comment. + # The numbers are still in the job summary. + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + github-token: ${{ secrets.GITHUB_TOKEN }} + script: | + const fs = require('fs'); + const marker = ''; + const { changed } = JSON.parse(fs.readFileSync('report.json', 'utf8')); + + const comments = await github.paginate(github.rest.issues.listComments, { + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: context.issue.number, + per_page: 100, + }); + const existing = comments.find( + (comment) => comment.user?.login === 'github-actions[bot]' && comment.body?.includes(marker), + ); + + // Most pull requests don't change any code samples. Don't post a + // comment saying so unless an earlier push changed them. + if (!changed && !existing) { + core.info('SDK code line counts are unchanged and there is no earlier comment; skipping.'); + return; + } + + const body = `${marker}\n${fs.readFileSync('report.md', 'utf8')}`; + if (existing) { + await github.rest.issues.updateComment({ + owner: context.repo.owner, + repo: context.repo.repo, + comment_id: existing.id, + body, + }); + core.info(`Updated the Snipsync coverage comment (${existing.id}).`); + } else { + const { data: comment } = await github.rest.issues.createComment({ + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: context.issue.number, + body, + }); + core.info(`Created the Snipsync coverage comment (${comment.id}).`); + } diff --git a/AGENTS.md b/AGENTS.md index 898a9d9568..5582d933ff 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -156,6 +156,9 @@ Adding or moving pages usually requires: - Prefer code extracted from CI-enabled sample repos via [Snipsync](https://github.com/temporalio/snipsync). - Snippets are wrapped in `` / ``. Edit the **source repo** named inside the wrapper, then run `yarn snipsync`. +- `yarn report:snipsync-coverage` reports the share of SDK code lines that come from Snipsync. On a pull request, the + Snipsync Coverage workflow comments when the change moves that number. Replacing a synced block with hand-written code + lowers it. See [UTILITIES.md](./readme/UTILITIES.md#snipsync-coverage). ## Pull requests diff --git a/bin/code-blocks.js b/bin/code-blocks.js new file mode 100644 index 0000000000..a363d877b8 --- /dev/null +++ b/bin/code-blocks.js @@ -0,0 +1,109 @@ +// Finds the fenced code blocks on a docs page and records which ones Snipsync +// manages. Shared by bin/report-snipsync-coverage.js and the checkers that +// compile hand-written samples. + +// Snipsync wraps synced code in either an HTML comment or an MDX comment. +const SNIPSTART = /^\s*(?:/g, '').replace(/\{\/\*.*?\*\/\}/g, ''); + if (rest.includes(''; + if (rest.includes('{/*')) return '*/}'; + return null; +} + +function dedent(line, indent) { + let i = 0; + while (i < indent && (line[i] === ' ' || line[i] === '\t')) i++; + return line.slice(i); +} + +// Every fenced code block on a page, with the 1-based line of its first line +// of code and whether Snipsync manages it. Code inside a list item or a JSX +// component is indented to match its fence; that indentation is removed. +// +// A synced block also carries `origin`, the `owner/repo` from the source link +// Snipsync wrote above it, or null when the wrapper turns source links off. +function extractCodeBlocks(source) { + const lines = source.split('\n'); + const blocks = []; + let fence = null; + let snipsync = false; + let origin = null; + let comment = null; + + for (let i = 0; i < lines.length; i++) { + const line = lines[i]; + + if (fence) { + const close = line.match(FENCE_CLOSE); + if (close && close[1][0] === fence.marker[0] && close[1].length >= fence.marker.length) { + blocks.push({ + lang: fence.lang, + line: fence.line, + code: fence.body.join('\n'), + snipsync: fence.snipsync, + origin: fence.origin, + }); + fence = null; + } else { + fence.body.push(dedent(line, fence.indent)); + } + continue; + } + + if (comment) { + if (line.includes(comment)) comment = null; + continue; + } + + if (SNIPSTART.test(line)) { + snipsync = true; + origin = null; + continue; + } + if (SNIPEND.test(line)) { + snipsync = false; + origin = null; + continue; + } + + if (snipsync) { + const link = line.match(SOURCE_LINK); + if (link) { + origin = `${link[1]}/${link[2]}`; + continue; + } + } + + const open = line.match(FENCE_OPEN); + if (open) { + fence = { + marker: open[2], + indent: open[1].length, + lang: open[3].toLowerCase(), + line: i + 2, + body: [], + snipsync, + origin: snipsync ? origin : null, + }; + continue; + } + + comment = openComment(line); + } + + return blocks; +} + +module.exports = { SNIPSTART, SNIPEND, extractCodeBlocks }; diff --git a/bin/report-snipsync-coverage.js b/bin/report-snipsync-coverage.js new file mode 100644 index 0000000000..6f9cd61da4 --- /dev/null +++ b/bin/report-snipsync-coverage.js @@ -0,0 +1,562 @@ +#!/usr/bin/env node + +// Reports how much of the SDK code shown in docs/ comes from Snipsync rather +// than being written into the page by hand. Code that Snipsync pulls from a +// sample repository is compiled in that repository's CI; hand-written code is +// not, so this is the share of SDK code we know builds. +// +// The headline counts non-blank lines inside fenced code blocks tagged with an +// SDK language (Python, TypeScript/JavaScript, Go, Java, C#, Ruby, PHP, Rust). +// A second line counts every fenced block except mermaid diagrams, which +// includes shell commands and configuration that would never come from a +// sample repository. +// +// node bin/report-snipsync-coverage.js # the working tree +// node bin/report-snipsync-coverage.js --ref # docs/ at a commit +// node bin/report-snipsync-coverage.js --base # compare against a commit +// node bin/report-snipsync-coverage.js --history [--since YYYY-MM-DD] [--every day|week|month] [--branch ] +// +// Output is a plain-text report by default; add --json or --markdown. History +// is always CSV, oldest first, one row per period (the last commit in it). +// +// The numbers depend only on the files in docs/, so the trend is rebuilt from +// git history instead of being stored anywhere. +// +// Always exits 0 unless it can't run at all. This is a report, not a check. + +const fs = require('fs'); +const path = require('path'); +const { execFileSync } = require('child_process'); +const { extractCodeBlocks } = require('./code-blocks'); + +const DOCS_DIR = 'docs'; +const DOC_EXTENSIONS = new Set(['.md', '.mdx']); +const MAX_BUFFER = 1024 * 1024 * 1024; + +const SDK_LANGUAGES = { + py: 'Python', + python: 'Python', + ts: 'TypeScript', + typescript: 'TypeScript', + tsx: 'TypeScript', + js: 'TypeScript', + javascript: 'TypeScript', + jsx: 'TypeScript', + go: 'Go', + golang: 'Go', + java: 'Java', + cs: 'C#', + csharp: 'C#', + 'c#': 'C#', + dotnet: 'C#', + rb: 'Ruby', + ruby: 'Ruby', + php: 'PHP', + rs: 'Rust', + rust: 'Rust', +}; + +// Never counted, even in the all-languages total. +const EXCLUDED_LANGUAGES = new Set(['mermaid']); + +// --------------------------------------------------------------------------- +// Counting +// --------------------------------------------------------------------------- + +function sdkLanguage(tag) { + return SDK_LANGUAGES[tag] || null; +} + +// The section a page belongs to: its top-level directory under docs/, with +// develop/ split one level further so each SDK gets its own row. +function sectionOf(file) { + const parts = file.split('/'); + if (parts[0] === DOCS_DIR) parts.shift(); + if (parts.length === 1) return '(top level)'; + if (parts[0] === 'develop' && parts.length > 2) return `develop/${parts[1]}`; + return parts[0]; +} + +function countLines(code) { + return code.split('\n').filter((line) => line.trim() !== '').length; +} + +function emptyCount() { + return { synced: 0, inline: 0 }; +} + +function percent(count) { + const total = count.synced + count.inline; + return total === 0 ? null : (count.synced / total) * 100; +} + +// Tallies the code blocks in `files`, an array of { path, source }. +function tally(files) { + const result = { + files: files.length, + sdk: emptyCount(), + all: emptyCount(), + sections: {}, + languages: {}, + origins: {}, + }; + + for (const file of files) { + const section = sectionOf(file.path); + for (const block of extractCodeBlocks(file.source)) { + if (EXCLUDED_LANGUAGES.has(block.lang)) continue; + const lines = countLines(block.code); + if (lines === 0) continue; + + const key = block.snipsync ? 'synced' : 'inline'; + result.sections[section] ??= { sdk: emptyCount(), all: emptyCount() }; + result.all[key] += lines; + result.sections[section].all[key] += lines; + + const language = sdkLanguage(block.lang); + if (!language) continue; + result.sdk[key] += lines; + result.sections[section].sdk[key] += lines; + result.languages[language] ??= emptyCount(); + result.languages[language][key] += lines; + if (block.snipsync) { + const origin = block.origin || 'unattributed'; + result.origins[origin] = (result.origins[origin] || 0) + lines; + } + } + } + + return result; +} + +// --------------------------------------------------------------------------- +// Reading docs/ +// --------------------------------------------------------------------------- + +function git(args, options = {}) { + return execFileSync('git', args, { maxBuffer: MAX_BUFFER, ...options }); +} + +function isDocFile(file) { + return DOC_EXTENSIONS.has(path.extname(file)); +} + +// Tracked files plus new files that aren't ignored, read from disk. This is +// what a commit of the working tree would contain. +function readWorkingTree() { + return git(['ls-files', '-z', '--cached', '--others', '--exclude-standard', '--', DOCS_DIR]) + .toString('utf8') + .split('\0') + .filter((file) => file && isDocFile(file) && fs.existsSync(file)) + .sort() + .map((file) => ({ path: file, source: fs.readFileSync(file, 'utf8') })); +} + +// Splits `git cat-file --batch` output into the contents of each object. +function parseCatFileBatch(buffer) { + const contents = []; + let offset = 0; + while (offset < buffer.length) { + const newline = buffer.indexOf(0x0a, offset); + if (newline === -1) break; + const header = buffer.toString('utf8', offset, newline).split(' '); + if (header[1] === 'missing') throw new Error(`git object ${header[0]} is missing`); + const size = Number(header[2]); + contents.push(buffer.toString('utf8', newline + 1, newline + 1 + size)); + offset = newline + 1 + size + 1; + } + return contents; +} + +// docs/ as it was at `ref`, read from git objects without a checkout. +function readAtRef(ref) { + const entries = git(['ls-tree', '-r', '-z', ref, '--', DOCS_DIR]) + .toString('utf8') + .split('\0') + .filter(Boolean) + .map((entry) => { + const [meta, file] = entry.split('\t'); + const [mode, type, sha] = meta.split(' '); + return { mode, type, sha, path: file }; + }) + .filter((entry) => entry.type === 'blob' && entry.mode !== '120000' && isDocFile(entry.path)) + .sort((a, b) => a.path.localeCompare(b.path)); + + if (entries.length === 0) return []; + const contents = parseCatFileBatch( + git(['cat-file', '--batch'], { input: entries.map((entry) => entry.sha).join('\n') + '\n' }) + ); + return entries.map((entry, i) => ({ path: entry.path, source: contents[i] })); +} + +function resolveCommit(ref) { + return git(['rev-parse', '--verify', `${ref}^{commit}`]) + .toString('utf8') + .trim(); +} + +// --------------------------------------------------------------------------- +// History +// --------------------------------------------------------------------------- + +// The key of the period a commit date falls in. Weeks start on Monday (UTC). +function periodKey(isoDate, every) { + const date = new Date(isoDate); + if (every === 'day') return date.toISOString().slice(0, 10); + if (every === 'month') return date.toISOString().slice(0, 7); + const monday = new Date(Date.UTC(date.getUTCFullYear(), date.getUTCMonth(), date.getUTCDate())); + monday.setUTCDate(monday.getUTCDate() - ((monday.getUTCDay() + 6) % 7)); + return monday.toISOString().slice(0, 10); +} + +// The last commit in each period, oldest period first. `commits` is newest +// first, as git log prints it. +function samplePeriods(commits, every) { + const seen = new Set(); + const samples = []; + for (const commit of commits) { + const key = periodKey(commit.date, every); + if (seen.has(key)) continue; + seen.add(key); + samples.push({ ...commit, period: key }); + } + return samples.reverse(); +} + +function defaultBranch() { + for (const ref of ['origin/main', 'main']) { + try { + resolveCommit(ref); + return ref; + } catch { + // try the next one + } + } + return 'HEAD'; +} + +function history({ since, every, branch }) { + const args = ['log', '--first-parent', '--format=%H%x09%cI']; + if (since) args.push(`--since=${since}`); + args.push(branch); + const commits = git(args) + .toString('utf8') + .split('\n') + .filter(Boolean) + .map((line) => { + const [sha, date] = line.split('\t'); + return { sha, date }; + }); + + const rows = ['period,commit_date,commit,sdk_synced,sdk_inline,sdk_pct,all_pct,guides_pct,design_patterns_pct']; + for (const sample of samplePeriods(commits, every)) { + const result = tally(readAtRef(sample.sha)); + rows.push( + [ + sample.period, + sample.date.slice(0, 10), + sample.sha.slice(0, 12), + result.sdk.synced, + result.sdk.inline, + csvPercent(percent(result.sdk)), + csvPercent(percent(result.all)), + csvPercent(sectionPercent(result, 'guides')), + csvPercent(sectionPercent(result, 'design-patterns')), + ].join(',') + ); + } + return rows.join('\n'); +} + +function sectionPercent(result, section) { + const entry = result.sections[section]; + return entry ? percent(entry.sdk) : null; +} + +function csvPercent(value) { + return value === null ? '' : value.toFixed(2); +} + +// --------------------------------------------------------------------------- +// Formatting +// --------------------------------------------------------------------------- + +function formatNumber(n) { + return n.toLocaleString('en-US'); +} + +function formatPercent(value) { + return value === null ? 'n/a' : `${value.toFixed(1)}%`; +} + +// The change between two percentages as shown, so 15.0% → 15.1% reads +0.1 +// even when the unrounded change is smaller. +function formatDelta(before, after) { + if (before === null || after === null) return ''; + const delta = Number(after.toFixed(1)) - Number(before.toFixed(1)); + if (Math.abs(delta) < 0.05) return '±0.0 pts'; + return `${delta > 0 ? '+' : '−'}${Math.abs(delta).toFixed(1)} pts`; +} + +function headline(count) { + return `${formatPercent(percent(count))} (${formatNumber(count.synced)} of ${formatNumber( + count.synced + count.inline + )} lines)`; +} + +// Sections with any SDK code, most SDK code first. +function sdkSections(result) { + return Object.entries(result.sections) + .filter(([, entry]) => entry.sdk.synced + entry.sdk.inline > 0) + .sort(([a, x], [b, y]) => y.sdk.synced + y.sdk.inline - (x.sdk.synced + x.sdk.inline) || a.localeCompare(b)); +} + +function sortedCounts(counts) { + return Object.entries(counts).sort( + ([a, x], [b, y]) => y.synced + y.inline - (x.synced + x.inline) || a.localeCompare(b) + ); +} + +function sortedOrigins(origins) { + return Object.entries(origins).sort(([a, x], [b, y]) => y - x || a.localeCompare(b)); +} + +function markdownTable(header, rows) { + const lines = [`| ${header.join(' | ')} |`, `| ${header.map((_, i) => (i === 0 ? '---' : '--:')).join(' | ')} |`]; + for (const row of rows) lines.push(`| ${row.join(' | ')} |`); + return lines.join('\n'); +} + +function textTable(header, rows) { + const widths = header.map((h, i) => Math.max(h.length, ...rows.map((row) => String(row[i]).length))); + const format = (row) => + row.map((cell, i) => (i === 0 ? String(cell).padEnd(widths[i]) : String(cell).padStart(widths[i]))).join(' '); + return [format(header), widths.map((w) => '-'.repeat(w)).join(' '), ...rows.map(format)].join('\n'); +} + +function countRow(name, count) { + return [name, formatNumber(count.synced), formatNumber(count.inline), formatPercent(percent(count))]; +} + +const COUNT_HEADER = ['Snipsync lines', 'Inline lines', 'Snipsync %']; + +function breakdownTables(result, table) { + return [ + table( + ['Section', ...COUNT_HEADER], + sdkSections(result).map(([name, entry]) => countRow(name, entry.sdk)) + ), + table( + ['Language', ...COUNT_HEADER], + sortedCounts(result.languages).map(([name, count]) => countRow(name, count)) + ), + table( + ['Snipsync origin', 'Lines'], + sortedOrigins(result.origins).map(([name, lines]) => [name, formatNumber(lines)]) + ), + ]; +} + +function formatText(result, label) { + const [sections, languages, origins] = breakdownTables(result, textTable); + return [ + `Snipsync coverage of SDK code in docs/${label ? ` at ${label}` : ''}`, + '', + `SDK languages: ${headline(result.sdk)}`, + `All languages: ${headline(result.all)}`, + '', + sections, + '', + languages, + '', + origins, + ].join('\n'); +} + +function formatMarkdown(result, label) { + const [sections, languages, origins] = breakdownTables(result, markdownTable); + return [ + '### Snipsync coverage', + '', + `**${formatPercent(percent(result.sdk))}** of SDK code lines in \`docs/\`${label ? ` at \`${label}\`` : ''} come from Snipsync (${formatNumber( + result.sdk.synced + )} of ${formatNumber(result.sdk.synced + result.sdk.inline)}). All languages: ${headline(result.all)}.`, + '', + sections, + '', + '
By language and by origin', + '', + languages, + '', + origins, + '', + '
', + ].join('\n'); +} + +function changedSections(base, head) { + const names = new Set([...Object.keys(base.sections), ...Object.keys(head.sections)]); + return [...names] + .filter((name) => { + const before = base.sections[name]?.sdk || emptyCount(); + const after = head.sections[name]?.sdk || emptyCount(); + return before.synced !== after.synced || before.inline !== after.inline; + }) + .sort(); +} + +function isChanged(base, head) { + return base.sdk.synced !== head.sdk.synced || base.sdk.inline !== head.sdk.inline; +} + +function arrow(before, after) { + return before === after ? formatNumber(after) : `${formatNumber(before)} → ${formatNumber(after)}`; +} + +function comparisonRows(base, head) { + return changedSections(base, head).map((name) => { + const before = base.sections[name]?.sdk || emptyCount(); + const after = head.sections[name]?.sdk || emptyCount(); + return [ + name, + arrow(before.synced, after.synced), + arrow(before.inline, after.inline), + `${formatPercent(percent(before))} → ${formatPercent(percent(after))}`, + formatDelta(percent(before), percent(after)), + ]; + }); +} + +const COMPARISON_HEADER = ['Section', 'Snipsync lines', 'Inline lines', 'Snipsync %', 'Change']; + +function formatComparisonText(base, head, labels) { + const rows = comparisonRows(base, head); + return [ + `Snipsync coverage of SDK code in docs/: ${labels.base} → ${labels.head}`, + '', + `SDK languages: ${headline(base.sdk)} → ${headline(head.sdk)} ${formatDelta(percent(base.sdk), percent(head.sdk))}`, + `All languages: ${headline(base.all)} → ${headline(head.all)}`, + '', + rows.length ? textTable(COMPARISON_HEADER, rows) : 'No section changed.', + '', + formatText(head, labels.head), + ].join('\n'); +} + +function formatComparisonMarkdown(base, head, labels) { + const rows = comparisonRows(base, head); + const [sections, languages, origins] = breakdownTables(head, markdownTable); + return [ + '### Snipsync coverage', + '', + `SDK code lines in \`docs/\` that come from Snipsync: **${formatPercent(percent(base.sdk))} → ${formatPercent( + percent(head.sdk) + )}** (${formatDelta(percent(base.sdk), percent(head.sdk))}), ${formatNumber(head.sdk.synced)} of ${formatNumber( + head.sdk.synced + head.sdk.inline + )} lines.`, + '', + rows.length ? markdownTable(COMPARISON_HEADER, rows) : 'No section changed.', + '', + '
All sections', + '', + `All languages: ${headline(head.all)}.`, + '', + sections, + '', + languages, + '', + origins, + '', + '
', + '', + `Compared with \`${labels.base}\`. Counts non-blank lines in fenced code blocks tagged with an SDK language. Hand-written code isn't compiled anywhere; code Snipsync pulls from a sample repository is built in that repository's CI. See [UTILITIES.md](https://github.com/temporalio/documentation/blob/main/readme/UTILITIES.md#snipsync-coverage).`, + ].join('\n'); +} + +// --------------------------------------------------------------------------- +// Command line +// --------------------------------------------------------------------------- + +// A full commit SHA is shortened for display; branch names and the like are +// shown as given. +function refLabel(ref) { + return /^[0-9a-f]{40}$/.test(ref) ? ref.slice(0, 7) : ref; +} + +function parseArgs(argv) { + const options = { format: 'text', every: 'week' }; + const valued = { '--ref': 'ref', '--base': 'base', '--since': 'since', '--every': 'every', '--branch': 'branch' }; + for (let i = 0; i < argv.length; i++) { + const arg = argv[i]; + if (arg === '--json') options.format = 'json'; + else if (arg === '--markdown') options.format = 'markdown'; + else if (arg === '--history') options.history = true; + else if (valued[arg]) { + if (i + 1 >= argv.length) throw new Error(`${arg} needs a value`); + options[valued[arg]] = argv[++i]; + } else throw new Error(`Unknown argument: ${arg}`); + } + if (!['day', 'week', 'month'].includes(options.every)) { + throw new Error(`--every must be day, week, or month, not ${options.every}`); + } + return options; +} + +function main(argv = process.argv.slice(2)) { + const options = parseArgs(argv); + + if (options.history) { + return history({ since: options.since, every: options.every, branch: options.branch || defaultBranch() }); + } + + const head = options.ref ? tally(readAtRef(resolveCommit(options.ref))) : tally(readWorkingTree()); + const headLabel = options.ref ? refLabel(options.ref) : 'working tree'; + + if (!options.base) { + if (options.format === 'json') return JSON.stringify({ ref: options.ref || null, ...head }, null, 2); + if (options.format === 'markdown') return formatMarkdown(head, options.ref && headLabel); + return formatText(head, options.ref && headLabel); + } + + const base = tally(readAtRef(resolveCommit(options.base))); + const labels = { base: refLabel(options.base), head: headLabel }; + if (options.format === 'json') { + return JSON.stringify( + { + base: { ref: options.base, ...base }, + head: { ref: options.ref || null, ...head }, + changed: isChanged(base, head), + }, + null, + 2 + ); + } + if (options.format === 'markdown') return formatComparisonMarkdown(base, head, labels); + return formatComparisonText(base, head, labels); +} + +module.exports = { + SDK_LANGUAGES, + sdkLanguage, + sectionOf, + countLines, + percent, + tally, + parseCatFileBatch, + periodKey, + samplePeriods, + changedSections, + isChanged, + formatDelta, + refLabel, + parseArgs, +}; + +if (require.main === module) { + try { + console.log(main()); + } catch (error) { + console.error(error.message); + process.exit(1); + } +} diff --git a/bin/report-snipsync-coverage.test.js b/bin/report-snipsync-coverage.test.js new file mode 100644 index 0000000000..a2b020b00e --- /dev/null +++ b/bin/report-snipsync-coverage.test.js @@ -0,0 +1,292 @@ +const { describe, it } = require('node:test'); +const assert = require('node:assert'); +const { extractCodeBlocks } = require('./code-blocks.js'); +const { + sdkLanguage, + sectionOf, + countLines, + percent, + tally, + parseCatFileBatch, + periodKey, + samplePeriods, + changedSections, + isChanged, + formatDelta, + refLabel, + parseArgs, +} = require('./report-snipsync-coverage.js'); + +const fence = '```'; + +describe('extractCodeBlocks', () => { + it('finds fences indented inside a TabItem and removes the indentation', () => { + const source = ['', '', ` ${fence}go`, ' func A() {}', ` ${fence}`, '', ''].join( + '\n' + ); + const [block] = extractCodeBlocks(source); + assert.strictEqual(block.lang, 'go'); + assert.strictEqual(block.code, 'func A() {}'); + assert.strictEqual(block.snipsync, false); + }); + + it('reads tilde fences and longer fences that contain shorter ones', () => { + const source = ['~~~python', 'x = 1', '~~~', '````md', fence + 'go', 'func A() {}', fence, '````'].join('\n'); + const blocks = extractCodeBlocks(source); + assert.deepStrictEqual( + blocks.map((b) => b.lang), + ['python', 'md'] + ); + assert.strictEqual(blocks[1].code, [fence + 'go', 'func A() {}', fence].join('\n')); + }); + + it('marks blocks inside an HTML-comment Snipsync wrapper and records the origin', () => { + const source = [ + '', + '[worker/main.go](https://github.com/temporalio/samples-go/blob/main/worker/main.go)', + fence + 'go', + 'package main', + fence, + '', + ].join('\n'); + const [block] = extractCodeBlocks(source); + assert.strictEqual(block.snipsync, true); + assert.strictEqual(block.origin, 'temporalio/samples-go'); + assert.strictEqual(block.code, 'package main'); + }); + + it('marks blocks inside an MDX-comment Snipsync wrapper with options', () => { + const source = [ + ' {/* SNIPSTART ts-client {"highlightedLines": "1-2"} */}', + ' [src/client.ts](https://github.com/temporalio/samples-typescript/blob/main/src/client.ts)', + ' ' + fence + 'ts {1-2}', + ' const a = 1;', + ' ' + fence, + ' {/* SNIPEND */}', + fence + 'ts', + 'const b = 2;', + fence, + ].join('\n'); + const blocks = extractCodeBlocks(source); + assert.strictEqual(blocks[0].snipsync, true); + assert.strictEqual(blocks[0].origin, 'temporalio/samples-typescript'); + assert.strictEqual(blocks[1].snipsync, false); + assert.strictEqual(blocks[1].origin, null); + }); + + it('leaves the origin null when the wrapper turns source links off', () => { + const source = [ + '', + fence + 'ts', + 'export {};', + fence, + '', + ].join('\n'); + const [block] = extractCodeBlocks(source); + assert.strictEqual(block.snipsync, true); + assert.strictEqual(block.origin, null); + }); + + it('skips fences inside an unclosed HTML comment', () => { + const source = ['', fence + 'go', 'func B() {}', fence].join('\n'); + const blocks = extractCodeBlocks(source); + assert.strictEqual(blocks.length, 1); + assert.strictEqual(blocks[0].code, 'func B() {}'); + }); +}); + +describe('sdkLanguage', () => { + it('normalizes SDK language tags', () => { + assert.strictEqual(sdkLanguage('py'), 'Python'); + assert.strictEqual(sdkLanguage('typescript'), 'TypeScript'); + assert.strictEqual(sdkLanguage('js'), 'TypeScript'); + assert.strictEqual(sdkLanguage('csharp'), 'C#'); + assert.strictEqual(sdkLanguage('rb'), 'Ruby'); + }); + + it('returns null for languages that are not SDK code', () => { + for (const tag of ['bash', 'yaml', 'json', 'mermaid', 'text', '']) { + assert.strictEqual(sdkLanguage(tag), null, tag); + } + }); +}); + +describe('sectionOf', () => { + it('uses the top-level directory under docs/', () => { + assert.strictEqual(sectionOf('docs/guides/lock-shared-resources.mdx'), 'guides'); + assert.strictEqual(sectionOf('docs/design-patterns/saga-pattern.mdx'), 'design-patterns'); + assert.strictEqual(sectionOf('docs/cloud/metrics/index.mdx'), 'cloud'); + }); + + it('splits develop/ by SDK directory', () => { + assert.strictEqual(sectionOf('docs/develop/go/workers/run.mdx'), 'develop/go'); + assert.strictEqual(sectionOf('docs/develop/plugins-guide.mdx'), 'develop'); + }); + + it('groups pages directly under docs/', () => { + assert.strictEqual(sectionOf('docs/glossary.md'), '(top level)'); + }); +}); + +describe('countLines', () => { + it('counts non-blank lines', () => { + assert.strictEqual(countLines('a\n\n \nb'), 2); + assert.strictEqual(countLines(''), 0); + }); +}); + +describe('tally', () => { + const page = [ + '', + '[a.go](https://github.com/temporalio/samples-go/blob/main/a.go)', + fence + 'go', + 'package a', + '', + 'func A() {}', + fence, + '', + fence + 'python', + 'x = 1', + fence, + fence + 'bash', + 'temporal server start-dev', + fence, + fence + 'mermaid', + 'flowchart TD', + fence, + ].join('\n'); + + it('counts synced and inline lines, excluding the source link and mermaid', () => { + const result = tally([{ path: 'docs/guides/a.mdx', source: page }]); + assert.deepStrictEqual(result.sdk, { synced: 2, inline: 1 }); + assert.deepStrictEqual(result.all, { synced: 2, inline: 2 }); + assert.deepStrictEqual(result.sections.guides.sdk, { synced: 2, inline: 1 }); + assert.deepStrictEqual(result.languages.Go, { synced: 2, inline: 0 }); + assert.deepStrictEqual(result.languages.Python, { synced: 0, inline: 1 }); + assert.deepStrictEqual(result.origins, { 'temporalio/samples-go': 2 }); + }); + + it('records synced blocks without a source link as unattributed', () => { + const source = ['', fence + 'go', 'func A() {}', fence].join('\n'); + const result = tally([{ path: 'docs/develop/go/a.mdx', source }]); + assert.deepStrictEqual(result.origins, { unattributed: 1 }); + }); +}); + +describe('percent', () => { + it('returns null when there is nothing to count', () => { + assert.strictEqual(percent({ synced: 0, inline: 0 }), null); + assert.strictEqual(percent({ synced: 1, inline: 3 }), 25); + }); +}); + +describe('parseCatFileBatch', () => { + it('splits objects by their byte size, including multi-byte characters', () => { + const first = 'café →'; + const second = 'plain\ntext'; + const buffer = Buffer.concat([ + Buffer.from(`aaa blob ${Buffer.byteLength(first)}\n`), + Buffer.from(first), + Buffer.from('\n'), + Buffer.from(`bbb blob ${Buffer.byteLength(second)}\n`), + Buffer.from(second), + Buffer.from('\n'), + ]); + assert.deepStrictEqual(parseCatFileBatch(buffer), [first, second]); + }); + + it('throws on a missing object', () => { + assert.throws(() => parseCatFileBatch(Buffer.from('abc missing\n')), /missing/); + }); +}); + +describe('periodKey', () => { + it('starts weeks on Monday', () => { + assert.strictEqual(periodKey('2026-10-04T23:00:00Z', 'week'), '2026-09-28'); + assert.strictEqual(periodKey('2026-10-05T01:00:00Z', 'week'), '2026-10-05'); + }); + + it('uses UTC dates for days and months', () => { + assert.strictEqual(periodKey('2026-10-05T23:30:00-07:00', 'day'), '2026-10-06'); + assert.strictEqual(periodKey('2026-09-30T12:00:00Z', 'month'), '2026-09'); + }); +}); + +describe('samplePeriods', () => { + it('keeps the newest commit in each period, oldest period first', () => { + const commits = [ + { sha: 'c', date: '2026-10-02T00:00:00Z' }, + { sha: 'b', date: '2026-10-01T00:00:00Z' }, + { sha: 'a', date: '2026-09-15T00:00:00Z' }, + ]; + assert.deepStrictEqual( + samplePeriods(commits, 'month').map((s) => [s.period, s.sha]), + [ + ['2026-09', 'a'], + ['2026-10', 'c'], + ] + ); + }); +}); + +describe('comparison', () => { + const base = { + sdk: { synced: 10, inline: 90 }, + sections: { guides: { sdk: { synced: 0, inline: 50 } }, cloud: { sdk: { synced: 10, inline: 40 } } }, + }; + + it('finds the sections whose SDK counts changed', () => { + const head = { + sdk: { synced: 30, inline: 70 }, + sections: { guides: { sdk: { synced: 20, inline: 30 } }, cloud: { sdk: { synced: 10, inline: 40 } } }, + }; + assert.deepStrictEqual(changedSections(base, head), ['guides']); + assert.strictEqual(isChanged(base, head), true); + }); + + it('reports a section that appears or disappears', () => { + const head = { sdk: base.sdk, sections: { cloud: base.sections.cloud } }; + assert.deepStrictEqual(changedSections(base, head), ['guides']); + }); + + it('is unchanged when only other languages change', () => { + assert.strictEqual(isChanged(base, { ...base, all: { synced: 1, inline: 1 } }), false); + }); +}); + +describe('formatDelta', () => { + it('formats percentage-point changes', () => { + assert.strictEqual(formatDelta(10, 12.34), '+2.3 pts'); + assert.strictEqual(formatDelta(12, 10), '−2.0 pts'); + assert.strictEqual(formatDelta(10, 10.01), '±0.0 pts'); + assert.strictEqual(formatDelta(null, 10), ''); + }); + + it('matches the rounded percentages it sits next to', () => { + assert.strictEqual(formatDelta(15.024, 15.072), '+0.1 pts'); + assert.strictEqual(formatDelta(15.04, 15.01), '±0.0 pts'); + }); +}); + +describe('refLabel', () => { + it('shortens full SHAs only', () => { + assert.strictEqual(refLabel('1c2c9d2bc555aaaaaaaaaaaaaaaaaaaaaaaaaaaa'), '1c2c9d2'); + assert.strictEqual(refLabel('origin/main'), 'origin/main'); + }); +}); + +describe('parseArgs', () => { + it('reads formats and valued options', () => { + assert.deepStrictEqual(parseArgs(['--base', 'main', '--markdown']), { + format: 'markdown', + every: 'week', + base: 'main', + }); + }); + + it('rejects unknown arguments, missing values, and bad periods', () => { + assert.throws(() => parseArgs(['--nope']), /Unknown argument/); + assert.throws(() => parseArgs(['--ref']), /needs a value/); + assert.throws(() => parseArgs(['--history', '--every', 'year']), /--every/); + }); +}); diff --git a/package.json b/package.json index ebcc6eeef7..979d2d4dcf 100644 --- a/package.json +++ b/package.json @@ -20,6 +20,7 @@ "check:metrics:sdks": "node ./bin/check-metrics-against-sdks.js", "check:orphans": "node ./bin/check-orphan-pages.js", "check:redirects": "node ./bin/validate-redirects.js", + "report:snipsync-coverage": "node ./bin/report-snipsync-coverage.js", "test": "node --test", "lint": "vale --output=JSON docs/**/**/*.mdx > vale-output.json; vale --output=JSON docs/**/**/*.md > vale-md-output.json", "lint:go": "vale docs/develop/go/*.mdx", diff --git a/readme/AUTOMATIONS.md b/readme/AUTOMATIONS.md index de1a421d51..d8769c311f 100644 --- a/readme/AUTOMATIONS.md +++ b/readme/AUTOMATIONS.md @@ -35,6 +35,7 @@ a 404 rather than a 403, so a scope mistake looks like a missing file. | Vale CI | Changes under `docs/` | No | `vale-ci.yml`, `.vale-ci.ini` | Runs only the rules in `.vale-ci.ini`, with `fail_on_error: false` and results limited to changed context. Treat those rules as the bar, not the full style set. See [AGENTS.md](../AGENTS.md#commands). | | Check Orphan Pages | Changes to docs or `sidebars.js` | No | `check-orphan-pages.yml`, `bin/check-orphan-pages.js` | Never fails; comments and annotates only. Accepted exceptions belong in `bin/orphan-pages-baseline.json` with a note. See [UTILITIES.md](./UTILITIES.md#check-orphan-pages). | | Check Metrics Against SDKs | Changes to the metrics page or its checker | No | `check-metrics-against-sdks.yml` | Drift reports but does not fail. On its scheduled runs it manages a tracking issue instead. | +| Snipsync Coverage | Changes under `docs/`, and pushes to `main` | No | `snipsync-coverage.yml`, `bin/report-snipsync-coverage.js` | Reports the share of SDK code lines that come from Snipsync. Comments on a pull request only when it changes that count, and writes the numbers to the job summary on `main`. Rebuild the trend with `--history`. See [UTILITIES.md](./UTILITIES.md#snipsync-coverage). | | Docs Preview Links | Every pull request | No | `docs-preview-links.yml`, `bin/generate-docs-preview-list.js` | Reads the preview URL out of Vercel's own comment, then upserts a single comment listing the pages the pull request changes. | | Visual Comparison | Pull requests labeled `visual-comparison` | No | `visual-comparison.yml` | Compares against baselines captured weekly by Screenshot Capture, and publishes its HTML report to a throwaway Vercel deployment linked from a pull request comment. Takes about 15 minutes. See [UTILITIES.md](./UTILITIES.md#visual-comparison). | | Dependabot | Weekly, for npm and GitHub Actions | No | `.github/dependabot.yml` | 14-day cooldown on new versions. | diff --git a/readme/UTILITIES.md b/readme/UTILITIES.md index a5c7690873..4faf504528 100644 --- a/readme/UTILITIES.md +++ b/readme/UTILITIES.md @@ -78,6 +78,39 @@ For each flagged file, decide whether it: This utility highlights potential issues — it's up to you to decide what belongs in our published documentation. +## snipsync-coverage + +`bin/report-snipsync-coverage.js` (run via `yarn report:snipsync-coverage`) reports how much of the SDK code in `docs/` +comes from Snipsync. Code that Snipsync pulls from a sample repository is built in that repository's CI. Code written +into a page by hand isn't compiled anywhere, so the percentage is the share of SDK code we know builds. + +The headline counts non-blank lines inside fenced code blocks tagged with an SDK language: Python, TypeScript or +JavaScript, Go, Java, C#, Ruby, PHP, and Rust. A second line counts every fenced block except mermaid diagrams. A block +counts as synced when it sits between `SNIPSTART` and `SNIPEND` markers, in either the `` or the `{/* */}` form. +The source link Snipsync writes above a block isn't counted as code, but its URL is how the report attributes synced +lines to a source repository. + +The report breaks the numbers down by section (the top-level directory under `docs/`, with `develop/` split by SDK), by +language, and by source repository. + +```bash +yarn report:snipsync-coverage # the working tree +yarn report:snipsync-coverage --ref # docs/ at a commit, without a checkout +yarn report:snipsync-coverage --base origin/main # what this branch changes +yarn report:snipsync-coverage --history --since 2026-01-01 --every month # the trend, as CSV +``` + +Add `--json` or `--markdown` to any of the first three. `--history` samples the last commit in each day, week (the +default), or month on the first-parent history of `origin/main` (or `--branch`), oldest first. + +The Snipsync Coverage workflow runs the comparison on every pull request that changes `docs/`, writes it to the job +summary, and keeps one PR comment up to date. It only comments when the SDK line counts change. On `main` it writes the +current numbers to the job summary. Nothing is stored. The numbers depend only on the files in `docs/`, so `--history` +rebuilds the trend whenever you need it. + +The code block parser lives in `bin/code-blocks.js`, so other scripts that need to tell synced code from hand-written +code can share it. + ## validate-redirects `bin/validate-redirects.js` (run via `yarn check:redirects`) checks the `redirects` in `vercel.json` for rules that