From c50e23b37a267632b6551451cba1b154b932e3c6 Mon Sep 17 00:00:00 2001 From: MK Date: Tue, 22 Sep 2026 15:52:42 +0800 Subject: [PATCH 1/4] bench: track config loading performance in CI --- .github/workflows/config-performance.yml | 122 ++++++++++ bench/config-performance/README.md | 35 +++ bench/config-performance/results.spec.ts | 62 +++++ bench/config-performance/results.ts | 102 ++++++++ bench/config-performance/run.ts | 282 +++++++++++++++++++++++ 5 files changed, 603 insertions(+) create mode 100644 .github/workflows/config-performance.yml create mode 100644 bench/config-performance/README.md create mode 100644 bench/config-performance/results.spec.ts create mode 100644 bench/config-performance/results.ts create mode 100644 bench/config-performance/run.ts diff --git a/.github/workflows/config-performance.yml b/.github/workflows/config-performance.yml new file mode 100644 index 0000000000..fe7c810fb4 --- /dev/null +++ b/.github/workflows/config-performance.yml @@ -0,0 +1,122 @@ +name: Config Performance + +permissions: + contents: read + actions: read + packages: read + +on: + pull_request: + types: [opened, synchronize, reopened] + paths: + - 'bench/config-performance/**' + - 'packages/**' + - 'crates/**' + - '.github/actions/**' + - '.github/workflows/config-performance.yml' + - '.node-version' + - '.cargo/**' + - 'Cargo.*' + - 'rust-toolchain.toml' + - 'pnpm-*.yaml' + - 'package.json' + push: + branches: [main] + paths: + - 'bench/config-performance/**' + - 'packages/**' + - 'crates/**' + - '.github/actions/**' + - '.github/workflows/config-performance.yml' + - '.node-version' + - '.cargo/**' + - 'Cargo.*' + - 'rust-toolchain.toml' + - 'pnpm-*.yaml' + - 'package.json' + schedule: + - cron: '23 2 * * *' + workflow_dispatch: + +concurrency: + group: config-performance-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.ref_name != 'main' }} + +defaults: + run: + shell: bash + +jobs: + benchmark: + name: Config loading benchmark + runs-on: namespace-profile-linux-x64-default + timeout-minutes: 30 + steps: + - uses: taiki-e/checkout-action@7d1e50e93dc4fb3bba58f85018fadf77898aee8b # v1.4.2 + - uses: ./.github/actions/clone + - uses: oxc-project/setup-rust@68c3199c5339f965e6e163924c3c450773eba42b # main + with: + save-cache: ${{ github.ref_name == 'main' }} + cache-key: config-performance + - uses: oxc-project/setup-node@1f1a5b4450c8905bd1830c3a908d22b960c4559a # main + - uses: ./.github/actions/build-upstream + with: + target: x86_64-unknown-linux-gnu + + # Prefer a successful main run. Before this workflow reaches main, use + # an earlier successful run of this PR so draft iterations get comparisons. + - name: Find previous measurements + id: baseline + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9 + with: + script: | + const repo = context.repo; + const current = await github.rest.actions.getWorkflowRun({ ...repo, run_id: context.runId }); + const { data } = await github.rest.actions.listWorkflowRuns({ + ...repo, workflow_id: current.data.workflow_id, status: 'success', per_page: 100, + }); + const runs = data.workflow_runs.filter(run => run.id !== context.runId); + const main = runs.filter(run => run.head_branch === 'main' && run.event !== 'pull_request'); + const pr = context.payload.pull_request; + const previousPR = pr ? runs.filter(run => + run.event === 'pull_request' && run.head_branch === pr.head.ref && + run.head_repository?.full_name === pr.head.repo.full_name + ) : []; + for (const run of [...main, ...previousPR]) { + const artifacts = await github.rest.actions.listWorkflowRunArtifacts({ ...repo, run_id: run.id }); + if (artifacts.data.artifacts.some(artifact => artifact.name === 'config-performance' && !artifact.expired)) { + core.setOutput('run-id', String(run.id)); + core.info(`Using baseline from ${run.html_url}`); + return; + } + } + core.info('No previous measurements; this run establishes a baseline.'); + + - name: Download previous measurements + if: steps.baseline.outputs.run-id != '' + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: config-performance + path: ${{ runner.temp }}/config-performance-baseline + run-id: ${{ steps.baseline.outputs.run-id }} + github-token: ${{ github.token }} + + - name: Run benchmark + env: + BASELINE_RUN: ${{ steps.baseline.outputs.run-id }} + run: | + args=() + if [[ -n "$BASELINE_RUN" ]]; then + args+=(--baseline "$RUNNER_TEMP/config-performance-baseline/results.json") + fi + node bench/config-performance/run.ts --output "$RUNNER_TEMP/config-performance" "${args[@]}" + + - name: Upload measurements + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: config-performance + path: ${{ runner.temp }}/config-performance + if-no-files-found: warn + retention-days: 90 + overwrite: true diff --git a/bench/config-performance/README.md b/bench/config-performance/README.md new file mode 100644 index 0000000000..64691baaa4 --- /dev/null +++ b/bench/config-performance/README.md @@ -0,0 +1,35 @@ +This benchmark tracks the config-loading costs described in [#2698](https://github.com/voidzero-dev/vite-plus/issues/2698). It runs the checkout's built CLI without modifying its code or dependencies. + +Build the checkout as described in [CONTRIBUTING.md](../../CONTRIBUTING.md), then run on Linux or macOS: + +```sh +node bench/config-performance/run.ts +``` + +Results go to `tmp/config-performance/results.json` and `summary.md`. To compare with an earlier result on the same machine and Node version: + +```sh +node bench/config-performance/run.ts --baseline previous-results.json +``` + +Use `--samples 2 --warmup 1` for a smoke run. A timing comparison requires at least seven samples. Defaults are 15 samples and three warmup rounds. + +The cases cover root and package-directory `vp check --fix`, missing and minimal configs, `defineConfig`, root `lint`/`fmt` blocks, standalone tools, and `vp staged` with a check or a no-op task. Each sample starts a fresh process. Cases rotate between rounds. File and Git preparation happen outside the timed interval, and each sample receives the same three unformatted TypeScript files. Staged samples verify the formatted Git index. Command errors and timeouts fail the benchmark. + +Config evaluations are measured in separate runs. Synchronous log writes do not affect the timing samples. The current ceilings are four evaluations for a root check, seven for a package check, and five for staged checks. Each Oxc child may evaluate its config at most once. These ceilings retain the current package-directory overhead until a separate optimization reduces it. Reduce the relevant ceiling with that optimization. + +The benchmark uses temporary projects outside the checkout so ancestor config discovery cannot find the repository's config. It runs the same Node executable in staged tasks and tool children. It uses a separate Node compile-cache directory and removes fixtures after completion. Warmups make this a measurement of fresh-process startup with warm filesystem and compile caches, not a cold-disk benchmark. + +The [Config Performance workflow](../../.github/workflows/config-performance.yml) runs on relevant PR updates, relevant pushes to `main`, daily, and on manual dispatch. Draft PRs run normally. The daily schedule becomes active after the workflow reaches `main`. + +CI retains JSON samples and Markdown reports for 90 days. It compares with the latest available successful `main` run. Before the workflow reaches `main`, PR updates use an earlier successful run of the same branch when available. The first run records a baseline. Workload, Node version, operating system, architecture, CPU model, and CPU count must match for a timing comparison; a mismatch is reported explicitly. + +A timing regression fails CI when all three conditions hold: + +- The median grows by more than 20%. +- The median grows by more than 40 ms. +- The current 25th percentile exceeds the baseline 75th percentile. + +This catches substantial, sustained regressions while tolerating isolated slow samples. It does not prove that smaller changes are harmless. Review the medians, p95 values, raw samples, and config counts when changing config resolution. CPU load and runner image changes can still affect results. Config-count ceilings apply even when timing results are not comparable. + +The workload hash changes with the fixture or case definitions, so those changes establish a new timing baseline. Tool versions are recorded but are not part of the compatibility check: dependency updates must remain visible in the comparison. diff --git a/bench/config-performance/results.spec.ts b/bench/config-performance/results.spec.ts new file mode 100644 index 0000000000..326b5f64f8 --- /dev/null +++ b/bench/config-performance/results.spec.ts @@ -0,0 +1,62 @@ +import { expect, test } from 'vite-plus/test'; + +import { compareReports, type BenchmarkReport } from './results.ts'; + +function report(samples: number[]): BenchmarkReport { + return { + schemaVersion: 1, + workload: 'fixture-v1', + revision: 'test-revision', + environment: { node: 'v22.18.0', platform: 'linux', arch: 'x64', cpu: 'test', cpus: 4 }, + versions: {}, + results: [{ id: 'root/check/minimal', samples, configLoads: [] }], + }; +} + +test('a single slow sample does not cause a timing regression', () => { + const baseline = report([490, 495, 499, 500, 501, 505, 510]); + const current = report([490, 495, 499, 500, 501, 505, 5000]); + expect(compareReports(current, baseline).regressions).toEqual([]); +}); + +test('a sustained slowdown fails the timing comparison', () => { + const baseline = report([490, 495, 499, 500, 501, 505, 510]); + const current = report([640, 645, 649, 650, 651, 655, 660]); + expect(compareReports(current, baseline).regressions).toHaveLength(1); +}); + +test('small absolute changes and overlapping distributions do not fail', () => { + expect( + compareReports( + report([130, 130, 130, 130, 130, 130, 130]), + report([100, 100, 100, 100, 100, 100, 100]), + ).regressions, + ).toEqual([]); + expect( + compareReports( + report([490, 500, 550, 650, 750, 800, 900]), + report([400, 450, 500, 500, 550, 700, 800]), + ).regressions, + ).toEqual([]); +}); + +test('different environments and workloads are explicitly not compared', () => { + const baseline = report([100, 100, 100, 100, 100, 100, 100]); + const current = report([500, 500, 500, 500, 500, 500, 500]); + current.environment.node = 'v24.0.0'; + expect(compareReports(current, baseline).markdown).toContain('comparison skipped'); + current.environment = baseline.environment; + current.workload = 'fixture-v2'; + expect(compareReports(current, baseline).markdown).toContain('comparison skipped'); +}); + +test('invalid or incomplete baseline data cannot silently pass', () => { + const current = report([500, 500, 500, 500, 500, 500, 500]); + expect(() => compareReports(current, report([Number.NaN, 1, 1, 1, 1, 1, 1]))).toThrow( + 'positive, finite', + ); + expect(() => compareReports(current, report([500]))).toThrow('seven samples'); + const baseline = report([500, 500, 500, 500, 500, 500, 500]); + baseline.results = []; + expect(() => compareReports(current, baseline)).toThrow('missing case'); +}); diff --git a/bench/config-performance/results.ts b/bench/config-performance/results.ts new file mode 100644 index 0000000000..98e51806dd --- /dev/null +++ b/bench/config-performance/results.ts @@ -0,0 +1,102 @@ +export interface CaseResult { + id: string; + samples: number[]; + configLoads: { role: string; count: number }[]; +} + +export interface BenchmarkReport { + schemaVersion: 1; + workload: string; + revision: string; + environment: { + node: string; + platform: string; + arch: string; + cpu: string; + cpus: number; + }; + versions: Record; + results: CaseResult[]; +} + +export function percentile(samples: number[], fraction: number): number { + if (samples.length === 0 || samples.some((sample) => !Number.isFinite(sample) || sample <= 0)) { + throw new Error('Timings must be a nonempty array of positive, finite numbers'); + } + const sorted = samples.toSorted((a, b) => a - b); + const index = (sorted.length - 1) * fraction; + const lower = Math.floor(index); + return sorted[lower] + (sorted[Math.ceil(index)] - sorted[lower]) * (index - lower); +} + +export function summarize(samples: number[]) { + return { + median: percentile(samples, 0.5), + p25: percentile(samples, 0.25), + p75: percentile(samples, 0.75), + p95: percentile(samples, 0.95), + }; +} + +export function compareReports(current: BenchmarkReport, baseline?: BenchmarkReport) { + const regressions: string[] = []; + let comparison = 'No baseline available; this run records the initial measurements.'; + const comparable = + baseline !== undefined && + baseline.schemaVersion === current.schemaVersion && + baseline.workload === current.workload && + JSON.stringify(baseline.environment) === JSON.stringify(current.environment); + + if (baseline && !comparable) { + comparison = 'Baseline environment or workload differs; timing comparison skipped.'; + } + if (comparable) { + comparison = `Compared with ${baseline.revision}.`; + } + const rows = current.results.map((result) => { + const stats = summarize(result.samples); + const previous = comparable + ? baseline.results.find((item) => item.id === result.id) + : undefined; + let change = '—'; + if (comparable && !previous) { + throw new Error(`Baseline is missing case ${result.id}`); + } + if (previous) { + if (previous.samples.length < 7 || result.samples.length < 7) { + throw new Error('A timing comparison requires at least seven samples per case'); + } + const before = summarize(previous.samples); + const delta = stats.median - before.median; + const ratio = stats.median / before.median - 1; + change = `${ratio >= 0 ? '+' : ''}${(ratio * 100).toFixed(1)}%`; + // Require a substantial slowdown across the distribution, not one outlier. + if (ratio > 0.2 && delta > 40 && stats.p25 > before.p75) { + regressions.push( + `${result.id}: ${before.median.toFixed(1)} → ${stats.median.toFixed(1)} ms (${change})`, + ); + } + } + const loads = + result.configLoads.map(({ role, count }) => `${role}: ${count}`).join(', ') || '0'; + return `| ${result.id} | ${stats.median.toFixed(1)} | ${stats.p95.toFixed(1)} | ${change} | ${loads} |`; + }); + const markdown = [ + '## Config performance', + '', + comparison, + '', + `Revision: \`${current.revision}\`. Node ${current.environment.node}; ${current.environment.platform}/${current.environment.arch}; ${current.environment.cpu}.`, + '', + '| Case | Median (ms) | p95 (ms) | Median change | Config evaluations (separate probe) |', + '| --- | ---: | ---: | ---: | --- |', + ...rows, + '', + 'Fresh processes, warm filesystem and Node compile caches. Config logging is disabled during timing.', + 'A timing regression requires >20% and >40 ms median growth, with current p25 above baseline p75.', + '', + ...regressions.map((regression) => `- Regression: ${regression}`), + '', + ].join('\n'); + return { regressions, markdown }; +} diff --git a/bench/config-performance/run.ts b/bench/config-performance/run.ts new file mode 100644 index 0000000000..11c161d20b --- /dev/null +++ b/bench/config-performance/run.ts @@ -0,0 +1,282 @@ +import { spawnSync } from 'node:child_process'; +import { createHash } from 'node:crypto'; +import { + appendFileSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + realpathSync, + rmSync, + symlinkSync, + writeFileSync, +} from 'node:fs'; +import { createRequire } from 'node:module'; +import { cpus, tmpdir } from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { fileURLToPath } from 'node:url'; +import { parseArgs } from 'node:util'; + +import { compareReports, type BenchmarkReport } from './results.ts'; + +const { values } = parseArgs({ + options: { + samples: { type: 'string', default: '15' }, + warmup: { type: 'string', default: '3' }, + output: { type: 'string', default: 'tmp/config-performance' }, + baseline: { type: 'string' }, + }, +}); +const samples = Number(values.samples); +const warmup = Number(values.warmup); +if (!Number.isInteger(samples) || samples < 1 || !Number.isInteger(warmup) || warmup < 1) { + throw new Error('--samples and --warmup must be positive integers'); +} +if (process.platform === 'win32') { + throw new Error('This benchmark currently supports Linux and macOS'); +} + +const repo = fileURLToPath(new URL('../../', import.meta.url)); +const cliPackage = path.join(repo, 'packages/cli'); +const cli = path.join(cliPackage, 'bin/vp'); +if (!existsSync(path.join(cliPackage, 'dist/bin.js'))) { + throw new Error('Build the checkout first: pnpm -F vite-plus build'); +} +const require = createRequire(path.join(cliPackage, 'package.json')); +const output = path.resolve(values.output); +mkdirSync(output, { recursive: true }); + +// Keep fixtures outside the checkout: Oxc ancestor discovery can find its config. +const temporary = realpathSync(mkdtempSync(path.join(tmpdir(), 'vp-config-performance-'))); +const fixture = path.join(temporary, 'project'); +const app = path.join(fixture, 'packages/app'); +const bin = path.join(temporary, 'bin'); +const log = path.join(temporary, 'config-loads.jsonl'); +const files = Array.from({ length: 3 }, (_, i) => `packages/app/src/f${i}.ts`); +const source = (i: number) => `export const f${i}={message:"fixture",value:${i}}\n`; +const configs = { + none: undefined, + minimal: "export default { staged: { '*.ts': 'vp check --fix' } };\n", + defined: + "import { defineConfig } from 'vite-plus';\nexport default defineConfig({ staged: { '*.ts': 'vp check --fix' } });\n", + blocks: "export default { staged: { '*.ts': 'vp check --fix' }, lint: {}, fmt: {} };\n", + noop: "export default { staged: { '*.ts': 'node -e \"process.exit(0)\"' } };\n", +}; +const cases = [ + { id: 'root/check/no-config', config: 'none', command: 'check', package: false, maxLoads: 0 }, + { id: 'root/check/minimal', config: 'minimal', command: 'check', package: false, maxLoads: 4 }, + { + id: 'root/check/defineConfig', + config: 'defined', + command: 'check', + package: false, + maxLoads: 4, + }, + { id: 'package/check/minimal', config: 'minimal', command: 'check', package: true, maxLoads: 7 }, + { id: 'package/check/blocks', config: 'blocks', command: 'check', package: true, maxLoads: 7 }, + { id: 'root/fmt/minimal', config: 'minimal', command: 'fmt', package: false, maxLoads: 1 }, + { id: 'root/lint/minimal', config: 'minimal', command: 'lint', package: false, maxLoads: 1 }, + { id: 'root/staged/minimal', config: 'minimal', command: 'staged', package: false, maxLoads: 5 }, + { id: 'root/staged/noop', config: 'noop', command: 'staged', package: false, maxLoads: 1 }, +] as const; + +const env = { ...process.env }; +for (const key of Object.keys(env)) { + if ( + key.startsWith('VP_') || + key.startsWith('GIT_') || + key === 'DEBUG' || + key === 'NODE_OPTIONS' + ) { + delete env[key]; + } +} +delete env.NODE_DISABLE_COMPILE_CACHE; +Object.assign(env, { + PATH: `${bin}${path.delimiter}${env.PATH ?? ''}`, + NODE_COMPILE_CACHE: path.join(temporary, 'compile-cache'), + NO_COLOR: '1', + FORCE_COLOR: '0', +}); + +function run(program: string, args: string[], cwd: string) { + const start = performance.now(); + const child = spawnSync(program, args, { cwd, env, encoding: 'utf8', timeout: 60_000 }); + const elapsed = performance.now() - start; + if (child.error || child.status !== 0) { + throw new Error( + `${program} ${args.join(' ')} failed (${child.status}, ${child.signal}):\n${child.error ?? ''}\n${child.stdout}\n${child.stderr}`, + ); + } + return { elapsed, stdout: child.stdout }; +} +const git = (...args: string[]) => + run('git', ['-c', 'core.hooksPath=/dev/null', ...args], fixture).stdout; + +function prepare(item: (typeof cases)[number], instrument: boolean) { + const config = configs[item.config]; + const configPath = path.join(fixture, 'vite.config.ts'); + rmSync(configPath, { force: true }); + if (config !== undefined) { + const probe = instrument + ? "import { appendFileSync } from 'node:fs';\nappendFileSync(process.env.VP_BENCH_CONFIG_LOG, JSON.stringify({ pid: process.pid, argv: process.argv }) + '\\n');\n" + : ''; + writeFileSync(configPath, probe + config); + } + for (const [i, file] of files.entries()) { + writeFileSync(path.join(fixture, file), source(i)); + } + if (item.command === 'staged') { + git('add', '--', ...files); + } + if (instrument) { + writeFileSync(log, ''); + env.VP_BENCH_CONFIG_LOG = log; + } else { + delete env.VP_BENCH_CONFIG_LOG; + } +} + +function measure(item: (typeof cases)[number]) { + const args: string[] = [cli, item.command]; + if (item.command === 'check' || item.command === 'lint') { + args.push('--fix'); + } + if (item.command !== 'staged') { + args.push(...files.map((file) => (item.package ? path.relative('packages/app', file) : file))); + } + const { elapsed } = run(process.execPath, args, item.package ? app : fixture); + // A fast no-op or an empty staged-file set is not a successful benchmark. + if (item.command !== 'lint' && item.config !== 'noop') { + for (const file of files) { + const formatted = + item.command === 'staged' + ? git('show', `:${file}`) + : readFileSync(path.join(fixture, file), 'utf8'); + if (!formatted.includes('= {')) { + throw new Error(`${item.id} did not format ${file}`); + } + } + } + return elapsed; +} + +const report: BenchmarkReport = { + schemaVersion: 1, + workload: createHash('sha256') + .update(JSON.stringify({ configs, cases, source: files.map((_, i) => source(i)) })) + .digest('hex'), + revision: run('git', ['rev-parse', 'HEAD'], repo).stdout.trim(), + environment: { + node: process.version, + platform: process.platform, + arch: process.arch, + cpu: cpus()[0].model, + cpus: cpus().length, + }, + versions: Object.fromEntries( + ['vite-plus', 'vite', 'oxlint', 'oxfmt'].map((name) => [ + name, + (require(`${name}/package.json`) as { version: string }).version, + ]), + ), + results: cases.map(({ id }) => ({ id, samples: [], configLoads: [] })), +}; + +try { + mkdirSync(path.join(app, 'src'), { recursive: true }); + mkdirSync(path.join(fixture, 'node_modules'), { recursive: true }); + mkdirSync(bin); + symlinkSync(cliPackage, path.join(fixture, 'node_modules/vite-plus'), 'dir'); + symlinkSync(cli, path.join(bin, 'vp')); + symlinkSync(process.execPath, path.join(bin, 'node')); + writeFileSync( + path.join(fixture, 'package.json'), + JSON.stringify({ private: true, type: 'module', workspaces: ['packages/*'] }), + ); + writeFileSync(path.join(fixture, 'pnpm-workspace.yaml'), 'packages:\n - packages/*\n'); + writeFileSync( + path.join(app, 'package.json'), + JSON.stringify({ name: 'config-perf-app', private: true, type: 'module' }), + ); + writeFileSync(path.join(fixture, '.gitignore'), 'node_modules\nvite.config.ts\n'); + for (const [i, file] of files.entries()) { + writeFileSync(path.join(fixture, file), `export const f${i} = ${i};\n`); + } + git('init', '--quiet'); + git('add', '.'); + git( + '-c', + 'user.name=Benchmark', + '-c', + 'user.email=benchmark@example.invalid', + '-c', + 'commit.gpgsign=false', + 'commit', + '--quiet', + '-m', + 'fixture', + ); + + // Separate diagnostic probes keep synchronous log writes out of timed configs. + for (const [index, item] of cases.entries()) { + prepare(item, true); + measure(item); + const entries = readFileSync(log, 'utf8') + .trim() + .split('\n') + .filter(Boolean) + .map((line) => JSON.parse(line) as { pid: number; argv: string[] }); + const processes = new Map(); + for (const entry of entries) { + const executable = path.basename(entry.argv[1]); + const role = executable === 'oxfmt' || executable === 'oxlint' ? executable : entry.argv[2]; + const current = processes.get(entry.pid) ?? { role, count: 0 }; + current.count++; + processes.set(entry.pid, current); + } + const loads = [...processes.values()]; + report.results[index].configLoads = loads; + if ( + entries.length > item.maxLoads || + loads.some(({ role, count }) => count > (role === 'check' && item.package ? 4 : 1)) + ) { + throw new Error(`${item.id}: config evaluation budget exceeded: ${JSON.stringify(loads)}`); + } + } + + // Rotate cases between rounds to spread thermal and runner-load changes. + for (let round = 0; round < warmup + samples; round++) { + for (let offset = 0; offset < cases.length; offset++) { + const index = (round + offset) % cases.length; + prepare(cases[index], false); + const elapsed = measure(cases[index]); + if (round >= warmup) { + report.results[index].samples.push(elapsed); + } + } + console.log( + `Completed ${round < warmup ? 'warmup' : 'sample'} round ${round + 1}/${warmup + samples}`, + ); + } + const baseline = values.baseline + ? (JSON.parse(readFileSync(values.baseline, 'utf8')) as BenchmarkReport) + : undefined; + const { regressions, markdown } = compareReports(report, baseline); + writeFileSync(path.join(output, 'results.json'), `${JSON.stringify(report, null, 2)}\n`); + writeFileSync(path.join(output, 'summary.md'), markdown); + if (process.env.GITHUB_STEP_SUMMARY) { + appendFileSync(process.env.GITHUB_STEP_SUMMARY, markdown); + } + console.log(markdown); + if (regressions.length > 0) { + process.exitCode = 1; + } +} catch (error) { + writeFileSync(path.join(output, 'failure.txt'), String(error)); + writeFileSync(path.join(output, 'results.json'), `${JSON.stringify(report, null, 2)}\n`); + throw error; +} finally { + rmSync(temporary, { recursive: true, force: true }); +} From c2702fdf0b640a00d95a70e42f74ae5ad4f1f95f Mon Sep 17 00:00:00 2001 From: MK Date: Tue, 22 Sep 2026 15:56:43 +0800 Subject: [PATCH 2/4] bench: preserve comparisons across count budgets and reruns --- .github/workflows/config-performance.yml | 12 +++++++++++- bench/config-performance/README.md | 4 ++-- bench/config-performance/results.ts | 4 +++- bench/config-performance/run.ts | 13 ++++++++++++- 4 files changed, 28 insertions(+), 5 deletions(-) diff --git a/.github/workflows/config-performance.yml b/.github/workflows/config-performance.yml index fe7c810fb4..8c31ff61d1 100644 --- a/.github/workflows/config-performance.yml +++ b/.github/workflows/config-performance.yml @@ -82,7 +82,17 @@ jobs: run.event === 'pull_request' && run.head_branch === pr.head.ref && run.head_repository?.full_name === pr.head.repo.full_name ) : []; - for (const run of [...main, ...previousPR]) { + // A rerun can compare with the retained artifact of its own last + // successful attempt. Upload replaces that artifact only at the end. + const previousAttempt = []; + if (current.data.run_attempt > 1) { + const attempt = await github.request( + 'GET /repos/{owner}/{repo}/actions/runs/{run_id}/attempts/{attempt_number}', + { ...repo, run_id: context.runId, attempt_number: current.data.run_attempt - 1 }, + ); + if (attempt.data.conclusion === 'success') previousAttempt.push(current.data); + } + for (const run of [...main, ...previousAttempt, ...previousPR]) { const artifacts = await github.rest.actions.listWorkflowRunArtifacts({ ...repo, run_id: run.id }); if (artifacts.data.artifacts.some(artifact => artifact.name === 'config-performance' && !artifact.expired)) { core.setOutput('run-id', String(run.id)); diff --git a/bench/config-performance/README.md b/bench/config-performance/README.md index 64691baaa4..43097b8453 100644 --- a/bench/config-performance/README.md +++ b/bench/config-performance/README.md @@ -22,7 +22,7 @@ The benchmark uses temporary projects outside the checkout so ancestor config di The [Config Performance workflow](../../.github/workflows/config-performance.yml) runs on relevant PR updates, relevant pushes to `main`, daily, and on manual dispatch. Draft PRs run normally. The daily schedule becomes active after the workflow reaches `main`. -CI retains JSON samples and Markdown reports for 90 days. It compares with the latest available successful `main` run. Before the workflow reaches `main`, PR updates use an earlier successful run of the same branch when available. The first run records a baseline. Workload, Node version, operating system, architecture, CPU model, and CPU count must match for a timing comparison; a mismatch is reported explicitly. +CI retains JSON samples and Markdown reports for 90 days. It compares with the latest available successful `main` run. Before the workflow reaches `main`, reruns can use their previous successful attempt, and PR updates can use an earlier successful run of the same branch. The first run records a baseline. Workload, Node version, operating system, architecture, CPU model, and CPU count must match for a timing comparison; a mismatch is reported explicitly. A timing regression fails CI when all three conditions hold: @@ -32,4 +32,4 @@ A timing regression fails CI when all three conditions hold: This catches substantial, sustained regressions while tolerating isolated slow samples. It does not prove that smaller changes are harmless. Review the medians, p95 values, raw samples, and config counts when changing config resolution. CPU load and runner image changes can still affect results. Config-count ceilings apply even when timing results are not comparable. -The workload hash changes with the fixture or case definitions, so those changes establish a new timing baseline. Tool versions are recorded but are not part of the compatibility check: dependency updates must remain visible in the comparison. +The workload hash changes with the fixture or command definitions, so those changes establish a new timing baseline. Evaluation ceilings are excluded from the hash so lowering a ceiling after an optimization preserves the timing comparison. Tool versions are recorded but are not part of the compatibility check: dependency updates must remain visible in the comparison. diff --git a/bench/config-performance/results.ts b/bench/config-performance/results.ts index 98e51806dd..2bef211de3 100644 --- a/bench/config-performance/results.ts +++ b/bench/config-performance/results.ts @@ -45,7 +45,9 @@ export function compareReports(current: BenchmarkReport, baseline?: BenchmarkRep baseline !== undefined && baseline.schemaVersion === current.schemaVersion && baseline.workload === current.workload && - JSON.stringify(baseline.environment) === JSON.stringify(current.environment); + (Object.keys(current.environment) as (keyof BenchmarkReport['environment'])[]).every( + (key) => baseline.environment[key] === current.environment[key], + ); if (baseline && !comparable) { comparison = 'Baseline environment or workload differs; timing comparison skipped.'; diff --git a/bench/config-performance/run.ts b/bench/config-performance/run.ts index 11c161d20b..d215407828 100644 --- a/bench/config-performance/run.ts +++ b/bench/config-performance/run.ts @@ -165,7 +165,18 @@ function measure(item: (typeof cases)[number]) { const report: BenchmarkReport = { schemaVersion: 1, workload: createHash('sha256') - .update(JSON.stringify({ configs, cases, source: files.map((_, i) => source(i)) })) + .update( + JSON.stringify({ + configs, + cases: cases.map(({ id, config, command, package: fromPackage }) => ({ + id, + config, + command, + fromPackage, + })), + source: files.map((_, i) => source(i)), + }), + ) .digest('hex'), revision: run('git', ['rev-parse', 'HEAD'], repo).stdout.trim(), environment: { From 00e4be77975bf3b39568d19776387ab9e62eb51a Mon Sep 17 00:00:00 2001 From: MK Date: Tue, 22 Sep 2026 16:04:20 +0800 Subject: [PATCH 3/4] ci: use the current attempt when selecting benchmark baselines --- .github/workflows/config-performance.yml | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/.github/workflows/config-performance.yml b/.github/workflows/config-performance.yml index 8c31ff61d1..2ad2b0ec4b 100644 --- a/.github/workflows/config-performance.yml +++ b/.github/workflows/config-performance.yml @@ -85,13 +85,15 @@ jobs: // A rerun can compare with the retained artifact of its own last // successful attempt. Upload replaces that artifact only at the end. const previousAttempt = []; - if (current.data.run_attempt > 1) { + const attemptNumber = Number(process.env.GITHUB_RUN_ATTEMPT); + if (attemptNumber > 1) { const attempt = await github.request( 'GET /repos/{owner}/{repo}/actions/runs/{run_id}/attempts/{attempt_number}', - { ...repo, run_id: context.runId, attempt_number: current.data.run_attempt - 1 }, + { ...repo, run_id: context.runId, attempt_number: attemptNumber - 1 }, ); if (attempt.data.conclusion === 'success') previousAttempt.push(current.data); } + core.info(`Attempt ${attemptNumber}: ${main.length} main, ${previousAttempt.length} previous attempt, ${previousPR.length} PR baseline candidates.`); for (const run of [...main, ...previousAttempt, ...previousPR]) { const artifacts = await github.rest.actions.listWorkflowRunArtifacts({ ...repo, run_id: run.id }); if (artifacts.data.artifacts.some(artifact => artifact.name === 'config-performance' && !artifact.expired)) { From a07ad93b9c076c0470d7c7fb616fe2ac84ec8a9c Mon Sep 17 00:00:00 2001 From: MK Date: Tue, 22 Sep 2026 17:24:21 +0800 Subject: [PATCH 4/4] ci: report config performance changes on pull requests --- .github/workflows/config-performance.yml | 98 ++++++++++++++++++------ bench/config-performance/README.md | 10 ++- bench/config-performance/results.spec.ts | 24 ++++++ bench/config-performance/results.ts | 25 ++++-- bench/config-performance/run.ts | 10 ++- 5 files changed, 131 insertions(+), 36 deletions(-) diff --git a/.github/workflows/config-performance.yml b/.github/workflows/config-performance.yml index 2ad2b0ec4b..28327834f1 100644 --- a/.github/workflows/config-performance.yml +++ b/.github/workflows/config-performance.yml @@ -20,27 +20,10 @@ on: - 'rust-toolchain.toml' - 'pnpm-*.yaml' - 'package.json' - push: - branches: [main] - paths: - - 'bench/config-performance/**' - - 'packages/**' - - 'crates/**' - - '.github/actions/**' - - '.github/workflows/config-performance.yml' - - '.node-version' - - '.cargo/**' - - 'Cargo.*' - - 'rust-toolchain.toml' - - 'pnpm-*.yaml' - - 'package.json' - schedule: - - cron: '23 2 * * *' - workflow_dispatch: concurrency: - group: config-performance-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: ${{ github.ref_name != 'main' }} + group: config-performance-${{ github.event.pull_request.number }} + cancel-in-progress: true defaults: run: @@ -51,20 +34,21 @@ jobs: name: Config loading benchmark runs-on: namespace-profile-linux-x64-default timeout-minutes: 30 + outputs: + report-ready: ${{ steps.measure.outputs.report-ready }} steps: - uses: taiki-e/checkout-action@7d1e50e93dc4fb3bba58f85018fadf77898aee8b # v1.4.2 - uses: ./.github/actions/clone - uses: oxc-project/setup-rust@68c3199c5339f965e6e163924c3c450773eba42b # main with: - save-cache: ${{ github.ref_name == 'main' }} + save-cache: false cache-key: config-performance - uses: oxc-project/setup-node@1f1a5b4450c8905bd1830c3a908d22b960c4559a # main - uses: ./.github/actions/build-upstream with: target: x86_64-unknown-linux-gnu - # Prefer a successful main run. Before this workflow reaches main, use - # an earlier successful run of this PR so draft iterations get comparisons. + # Prefer the base branch of a stacked PR, then an earlier run of this PR. - name: Find previous measurements id: baseline uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9 @@ -76,8 +60,11 @@ jobs: ...repo, workflow_id: current.data.workflow_id, status: 'success', per_page: 100, }); const runs = data.workflow_runs.filter(run => run.id !== context.runId); - const main = runs.filter(run => run.head_branch === 'main' && run.event !== 'pull_request'); const pr = context.payload.pull_request; + const baseBranch = pr.base.ref !== 'main' ? runs.filter(run => + run.event === 'pull_request' && run.head_branch === pr.base.ref && + run.head_repository?.full_name === pr.base.repo.full_name + ) : []; const previousPR = pr ? runs.filter(run => run.event === 'pull_request' && run.head_branch === pr.head.ref && run.head_repository?.full_name === pr.head.repo.full_name @@ -93,8 +80,8 @@ jobs: ); if (attempt.data.conclusion === 'success') previousAttempt.push(current.data); } - core.info(`Attempt ${attemptNumber}: ${main.length} main, ${previousAttempt.length} previous attempt, ${previousPR.length} PR baseline candidates.`); - for (const run of [...main, ...previousAttempt, ...previousPR]) { + core.info(`Attempt ${attemptNumber}: ${baseBranch.length} base branch, ${previousAttempt.length} previous attempt, ${previousPR.length} PR baseline candidates.`); + for (const run of [...baseBranch, ...previousAttempt, ...previousPR]) { const artifacts = await github.rest.actions.listWorkflowRunArtifacts({ ...repo, run_id: run.id }); if (artifacts.data.artifacts.some(artifact => artifact.name === 'config-performance' && !artifact.expired)) { core.setOutput('run-id', String(run.id)); @@ -114,6 +101,7 @@ jobs: github-token: ${{ github.token }} - name: Run benchmark + id: measure env: BASELINE_RUN: ${{ steps.baseline.outputs.run-id }} run: | @@ -132,3 +120,63 @@ jobs: if-no-files-found: warn retention-days: 90 overwrite: true + + comment: + name: Report config performance changes + needs: benchmark + # Report timing regressions even when the benchmark fails its timing gate. + # Fork tokens cannot write comments; their reports stay in the job summary. + if: >- + always() && !cancelled() && + needs.benchmark.outputs.report-ready == 'true' && + github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-latest + permissions: + actions: read + issues: write + pull-requests: write + steps: + - name: Download measurements + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: config-performance + path: ${{ runner.temp }}/config-performance + + - name: Update PR comment for changes beyond five percent + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9 + env: + REPORT_DIR: ${{ runner.temp }}/config-performance + with: + script: | + const fs = require('node:fs'); + const path = require('node:path'); + const pr = context.payload.pull_request; + const { data: latest } = await github.rest.pulls.get({ + ...context.repo, pull_number: pr.number, + }); + if (latest.state !== 'open' || latest.head.sha !== pr.head.sha) { + core.info('The PR closed or its head changed; skip this older report.'); + return; + } + const marker = ''; + const comments = await github.paginate(github.rest.issues.listComments, { + ...context.repo, issue_number: pr.number, + }); + const existing = comments.find(comment => + comment.user?.login === 'github-actions[bot]' && comment.body?.includes(marker) + ); + const report = fs.readFileSync(path.join(process.env.REPORT_DIR, 'comment.md'), 'utf8'); + if (!report) { + if (existing) { + await github.rest.issues.deleteComment({ ...context.repo, comment_id: existing.id }); + } + core.info('No comparable median change beyond ±5%; no PR comment needed.'); + return; + } + const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}/attempts/${process.env.GITHUB_RUN_ATTEMPT}`; + const body = `${marker}\n\n[Benchmark run](${runUrl}) for PR head \`${pr.head.sha}\`.\n\n${report}`; + if (existing) { + await github.rest.issues.updateComment({ ...context.repo, comment_id: existing.id, body }); + } else { + await github.rest.issues.createComment({ ...context.repo, issue_number: pr.number, body }); + } diff --git a/bench/config-performance/README.md b/bench/config-performance/README.md index 43097b8453..b2d4a801fe 100644 --- a/bench/config-performance/README.md +++ b/bench/config-performance/README.md @@ -6,7 +6,7 @@ Build the checkout as described in [CONTRIBUTING.md](../../CONTRIBUTING.md), the node bench/config-performance/run.ts ``` -Results go to `tmp/config-performance/results.json` and `summary.md`. To compare with an earlier result on the same machine and Node version: +Results go to `tmp/config-performance/results.json` and `summary.md`. The summary shows baseline → current values for median and p95, plus the absolute and percentage median change. To compare with an earlier result on the same machine and Node version: ```sh node bench/config-performance/run.ts --baseline previous-results.json @@ -20,9 +20,13 @@ Config evaluations are measured in separate runs. Synchronous log writes do not The benchmark uses temporary projects outside the checkout so ancestor config discovery cannot find the repository's config. It runs the same Node executable in staged tasks and tool children. It uses a separate Node compile-cache directory and removes fixtures after completion. Warmups make this a measurement of fresh-process startup with warm filesystem and compile caches, not a cold-disk benchmark. -The [Config Performance workflow](../../.github/workflows/config-performance.yml) runs on relevant PR updates, relevant pushes to `main`, daily, and on manual dispatch. Draft PRs run normally. The daily schedule becomes active after the workflow reaches `main`. +The [Config Performance workflow](../../.github/workflows/config-performance.yml) runs only when a relevant PR opens, updates, or reopens. Draft PRs run normally. -CI retains JSON samples and Markdown reports for 90 days. It compares with the latest available successful `main` run. Before the workflow reaches `main`, reruns can use their previous successful attempt, and PR updates can use an earlier successful run of the same branch. The first run records a baseline. Workload, Node version, operating system, architecture, CPU model, and CPU count must match for a timing comparison; a mismatch is reported explicitly. +CI retains JSON samples and Markdown reports for 90 days. Stacked PRs prefer measurements from their base branch. Otherwise, reruns can use their previous successful attempt, and PR updates can use an earlier successful run of the same branch. The first run records a baseline. Workload, Node version, operating system, architecture, CPU model, and CPU count must match for a timing comparison; a mismatch is reported explicitly. + +If any comparable case's median changes by more than +5% or −5%, CI posts or updates one PR comment with the full comparison and a link to the run. Faster and slower results both trigger a comment, without an absolute-time threshold. Exactly ±5% does not trigger a comment. CI removes its previous comment when no case exceeds this threshold or comparison is unavailable. Missing or incompatible baselines do not trigger comments. Fork PRs retain the report in the job summary because their GitHub token cannot write comments. + +The notification threshold is separate from the failure threshold below. A timing regression still produces a comment if the benchmark fails. `comment.md` contains the comment text, or is empty when no notification is needed. A timing regression fails CI when all three conditions hold: diff --git a/bench/config-performance/results.spec.ts b/bench/config-performance/results.spec.ts index 326b5f64f8..8528d51179 100644 --- a/bench/config-performance/results.spec.ts +++ b/bench/config-performance/results.spec.ts @@ -40,6 +40,30 @@ test('small absolute changes and overlapping distributions do not fail', () => { ).toEqual([]); }); +test('PR notifications include improvements and small absolute changes beyond five percent', () => { + const baseline = report(Array(7).fill(100)); + for (const median of [94, 106]) { + const comparison = compareReports(report(Array(7).fill(median)), baseline); + expect(comparison.notableChanges).toEqual(['root/check/minimal']); + expect(comparison.regressions).toEqual([]); + } +}); + +test('PR notifications exclude changes within or exactly at five percent', () => { + const baseline = report(Array(7).fill(100)); + for (const median of [95, 96, 100, 104, 105]) { + expect(compareReports(report(Array(7).fill(median)), baseline).notableChanges).toEqual([]); + } +}); + +test('PR notifications require a compatible baseline', () => { + const baseline = report(Array(7).fill(100)); + const current = report(Array(7).fill(500)); + expect(compareReports(current).notableChanges).toEqual([]); + current.environment.node = 'v24.0.0'; + expect(compareReports(current, baseline).notableChanges).toEqual([]); +}); + test('different environments and workloads are explicitly not compared', () => { const baseline = report([100, 100, 100, 100, 100, 100, 100]); const current = report([500, 500, 500, 500, 500, 500, 500]); diff --git a/bench/config-performance/results.ts b/bench/config-performance/results.ts index 2bef211de3..bfc2c7576c 100644 --- a/bench/config-performance/results.ts +++ b/bench/config-performance/results.ts @@ -40,6 +40,7 @@ export function summarize(samples: number[]) { export function compareReports(current: BenchmarkReport, baseline?: BenchmarkReport) { const regressions: string[] = []; + const notableChanges: string[] = []; let comparison = 'No baseline available; this run records the initial measurements.'; const comparable = baseline !== undefined && @@ -53,13 +54,15 @@ export function compareReports(current: BenchmarkReport, baseline?: BenchmarkRep comparison = 'Baseline environment or workload differs; timing comparison skipped.'; } if (comparable) { - comparison = `Compared with ${baseline.revision}.`; + comparison = `Baseline revision: \`${baseline.revision}\`.`; } const rows = current.results.map((result) => { const stats = summarize(result.samples); const previous = comparable ? baseline.results.find((item) => item.id === result.id) : undefined; + let baselineMedian = '—'; + let baselineP95 = '—'; let change = '—'; if (comparable && !previous) { throw new Error(`Baseline is missing case ${result.id}`); @@ -71,26 +74,34 @@ export function compareReports(current: BenchmarkReport, baseline?: BenchmarkRep const before = summarize(previous.samples); const delta = stats.median - before.median; const ratio = stats.median / before.median - 1; - change = `${ratio >= 0 ? '+' : ''}${(ratio * 100).toFixed(1)}%`; + baselineMedian = before.median.toFixed(1); + baselineP95 = before.p95.toFixed(1); + const percentage = `${ratio >= 0 ? '+' : ''}${(ratio * 100).toFixed(1)}%`; + change = `${delta >= 0 ? '+' : ''}${delta.toFixed(1)} ms (${percentage})`; + // Compare the delta directly so exactly ±5% does not cross the threshold + // through division rounding (for example, 105 / 100 - 1). + if (Math.abs(delta) > before.median * 0.05) { + notableChanges.push(result.id); + } // Require a substantial slowdown across the distribution, not one outlier. if (ratio > 0.2 && delta > 40 && stats.p25 > before.p75) { regressions.push( - `${result.id}: ${before.median.toFixed(1)} → ${stats.median.toFixed(1)} ms (${change})`, + `${result.id}: ${baselineMedian} → ${stats.median.toFixed(1)} ms (${percentage})`, ); } } const loads = result.configLoads.map(({ role, count }) => `${role}: ${count}`).join(', ') || '0'; - return `| ${result.id} | ${stats.median.toFixed(1)} | ${stats.p95.toFixed(1)} | ${change} | ${loads} |`; + return `| ${result.id} | ${baselineMedian} → ${stats.median.toFixed(1)} | ${baselineP95} → ${stats.p95.toFixed(1)} | ${change} | ${loads} |`; }); const markdown = [ '## Config performance', '', comparison, '', - `Revision: \`${current.revision}\`. Node ${current.environment.node}; ${current.environment.platform}/${current.environment.arch}; ${current.environment.cpu}.`, + `Current revision: \`${current.revision}\`. Node ${current.environment.node}; ${current.environment.platform}/${current.environment.arch}; ${current.environment.cpu}.`, '', - '| Case | Median (ms) | p95 (ms) | Median change | Config evaluations (separate probe) |', + '| Case | Median (ms), baseline → current | p95 (ms), baseline → current | Median change | Config evaluations (current, separate probe) |', '| --- | ---: | ---: | ---: | --- |', ...rows, '', @@ -100,5 +111,5 @@ export function compareReports(current: BenchmarkReport, baseline?: BenchmarkRep ...regressions.map((regression) => `- Regression: ${regression}`), '', ].join('\n'); - return { regressions, markdown }; + return { regressions, notableChanges, markdown }; } diff --git a/bench/config-performance/run.ts b/bench/config-performance/run.ts index d215407828..7615b991c8 100644 --- a/bench/config-performance/run.ts +++ b/bench/config-performance/run.ts @@ -274,9 +274,17 @@ try { const baseline = values.baseline ? (JSON.parse(readFileSync(values.baseline, 'utf8')) as BenchmarkReport) : undefined; - const { regressions, markdown } = compareReports(report, baseline); + const { regressions, notableChanges, markdown } = compareReports(report, baseline); writeFileSync(path.join(output, 'results.json'), `${JSON.stringify(report, null, 2)}\n`); writeFileSync(path.join(output, 'summary.md'), markdown); + const comment = + notableChanges.length > 0 + ? `${notableChanges.length} case(s) changed by more than ±5% in median time. Negative changes are faster; positive changes are slower.\n\n${markdown}` + : ''; + writeFileSync(path.join(output, 'comment.md'), comment); + if (process.env.GITHUB_OUTPUT) { + appendFileSync(process.env.GITHUB_OUTPUT, 'report-ready=true\n'); + } if (process.env.GITHUB_STEP_SUMMARY) { appendFileSync(process.env.GITHUB_STEP_SUMMARY, markdown); }