benchmark: add --analyze mode to compare.js

Add an --analyze flag that performs statistical analysis directly
after benchmarks complete, eliminating the need for R and compare.R.

When --analyze is specified, compare.js collects the rate data during
the run and prints a statistical summary table instead of CSV output.
The table matches the format of compare.R: improvement percentage,
significance stars (* p<0.05, ** p<0.01, *** p<0.001), and confidence
intervals at three risk levels.

Also adds a --max-regression N option that causes the compare.js to
exit with 1 (error) when the `--new` is N% slower. Useful for CI
use to detect regressions.

Uses the histogram API's welchTest() and cohensD() methods introduced
in the previous commit. Benchmark rates are scaled to integers for
HdrHistogram recording; the --scale option (default 1000) controls
the multiplier for precision.

Usage:
  node benchmark/compare.js --old ./node-old --new ./node-new \
    --analyze url

Signed-off-by: James M Snell <jasnell@gmail.com>
Assisted-by: Opencode/Opus
PR-URL: https://github.com/nodejs/node/pull/65416
Reviewed-By: Matteo Collina <matteo.collina@gmail.com>
Reviewed-By: Chengzhong Wu <legendecas@gmail.com>
This commit is contained in:
James M Snell committed 2026-08-23 20:29:20 +00:00
1 parent 524dee4372
commit bf67fdc6d7
3 files changed
+290 -28

No files matched your search

+6 -2
View File
@@ -25,9 +25,10 @@ function getTime(diff) {
// A run is an item in the job queue: { binary, filename, iter }
// A config is an item in the subqueue: { binary, filename, iter, configs }
class BenchmarkProgress {
constructor(queue, benchmarks) {
constructor(queue, benchmarks, options = {}) {
this.queue = queue; // Scheduled runs.
this.benchmarks = benchmarks; // Filenames of scheduled benchmarks.
this.analyze = !!options.analyze; // stdout is not piped, but unused.
this.completedRuns = 0; // Number of completed runs.
this.scheduledRuns = queue.length; // Number of scheduled runs.
// Time when starting to run benchmarks.
@@ -107,7 +108,10 @@ class BenchmarkProgress {
}
updateProgress() {
if (!process.stderr.isTTY || process.stdout.isTTY) {
// Progress renders on stderr when stdout is piped (not a TTY).
// In --analyze mode, stdout is the terminal but is unused during
// the run, so treat it the same as piped.
if (!process.stderr.isTTY || (process.stdout.isTTY && !this.analyze)) {
return;
}
readline.clearLine(process.stderr);
+230 -9
View File
@@ -13,7 +13,8 @@ const cli = new CLI(`usage: ./node compare.js [options] [--] <category> ...
Run each benchmark in the <category> directory many times using two different
node versions. More than one <category> directory can be specified.
The output is formatted as csv, which can be processed using for
example 'compare.R'.
example 'compare.R'. Use --analyze to perform statistical analysis
directly without R.
--new ./new-node-binary new node binary (required)
--old ./old-node-binary old node binary (required)
@@ -24,13 +25,21 @@ const cli = new CLI(`usage: ./node compare.js [options] [--] <category> ...
repeated)
--set variable=value set benchmark variable (can be repeated)
--no-progress don't show benchmark progress indicator
--analyze perform statistical analysis after benchmarks
complete (Welch's t-test, effect size) instead
of printing csv output
--scale 1000 rate-to-integer multiplier for histogram
precision when using --analyze (default: 1000)
--max-regression N exit with code 1 if any statistically
significant regression exceeds N% (implies
--analyze)
Examples:
--set CPUSET=0 Runs benchmarks on CPU core 0.
--set CPUSET=0-2 Specifies that benchmarks should run on CPU cores 0 to 2.
Note: The CPUSET format should match the specifications of the 'taskset' command
`, { arrayArgs: ['set', 'filter', 'exclude'], boolArgs: ['no-progress'] });
`, { arrayArgs: ['set', 'filter', 'exclude'], boolArgs: ['no-progress', 'analyze'] });
if (!cli.optional.new || !cli.optional.old) {
cli.abort(cli.usage);
@@ -38,6 +47,11 @@ if (!cli.optional.new || !cli.optional.old) {
const binaries = ['old', 'new'];
const runs = cli.optional.runs ? parseInt(cli.optional.runs, 10) : 30;
const maxRegression = cli.optional['max-regression'] ?
parseFloat(cli.optional['max-regression']) :
0;
const analyze = !!cli.optional.analyze || maxRegression > 0;
const scale = cli.optional.scale ? parseInt(cli.optional.scale, 10) : 1000;
const benchmarks = cli.benchmarks();
if (benchmarks.length === 0) {
@@ -46,6 +60,9 @@ if (benchmarks.length === 0) {
return;
}
// When --analyze is set, collect results for statistical analysis.
const results = analyze ? new Map() : null;
// Create queue from the benchmarks list such both node versions are tested
// `runs` amount of times each.
// Note: BenchmarkProgress relies on this order to estimate
@@ -61,15 +78,17 @@ for (const filename of benchmarks) {
}
// queue.length = binary.length * runs * benchmarks.length
// Print csv header
console.log('"binary","filename","configuration","rate","time"');
// Print csv header (unless analyzing inline).
if (!analyze) {
console.log('"binary","filename","configuration","rate","time"');
}
const kStartOfQueue = 0;
const showProgress = !cli.optional['no-progress'];
let progress;
if (showProgress) {
progress = new BenchmarkProgress(queue, benchmarks);
progress = new BenchmarkProgress(queue, benchmarks, { analyze });
progress.startQueue(kStartOfQueue);
}
@@ -99,11 +118,20 @@ if (showProgress) {
conf += ` ${key}=${inspect(data.conf[key])}`;
}
conf = conf.slice(1);
// Escape quotes (") for correct csv formatting
conf = conf.replace(/"/g, '""');
console.log(`"${job.binary}","${job.filename}","${conf}",` +
`${data.rate},${data.time}`);
if (analyze) {
// Collect results for post-run analysis.
const name = `${job.filename} ${conf}`;
if (!results.has(name)) {
results.set(name, { old: [], new: [] });
}
results.get(name)[job.binary].push(data.rate);
} else {
// Escape quotes (") for correct csv formatting
conf = conf.replace(/"/g, '""');
console.log(`"${job.binary}","${job.filename}","${conf}",` +
`${data.rate},${data.time}`);
}
if (showProgress) {
// One item in the subqueue has been completed.
progress.completeConfig(data);
@@ -125,6 +153,199 @@ if (showProgress) {
// If there are more benchmarks execute the next
if (i + 1 < queue.length) {
recursive(i + 1);
} else if (analyze) {
printAnalysis(results, scale, maxRegression);
}
});
})(kStartOfQueue);
function printAnalysis(results, scale, maxRegression) {
const { createHistogram } = require('node:perf_hooks');
// Build per-benchmark histograms and run statistical tests.
const rows = [];
let maxNameLen = 0;
let skipped = 0;
for (const [name, { old: oldRates, new: newRates }] of results) {
if (oldRates.length < 2 || newRates.length < 2) {
skipped++;
continue;
}
const hOld = createHistogram({ figures: 3 });
const hNew = createHistogram({ figures: 3 });
for (const r of oldRates) hOld.record(Math.max(1, Math.round(r * scale)));
for (const r of newRates) hNew.record(Math.max(1, Math.round(r * scale)));
const oldMean = oldRates.reduce((a, b) => a + b, 0) / oldRates.length;
const newMean = newRates.reduce((a, b) => a + b, 0) / newRates.length;
const improvement = ((newMean - oldMean) / oldMean) * 100;
// Query the three confidence levels. The p-value and t-statistic
// are the same regardless of the confidence level, so we extract
// them from the first result.
const w95 = hOld.welchTest(hNew, { confidence: 0.95 });
const w99 = hOld.welchTest(hNew, { confidence: 0.99 });
const w999 = hOld.welchTest(hNew, { confidence: 0.999 });
// Significance stars matching compare.R convention.
let stars = '';
if (w95.pValue < 0.001) stars = '***';
else if (w95.pValue < 0.01) stars = ' **';
else if (w95.pValue < 0.05) stars = ' *';
// Confidence intervals expressed as percentage of the old mean.
const ciPct = (w) => {
const half =
(w.confidenceInterval.upper - w.confidenceInterval.lower) / 2;
return (half / (oldMean * scale)) * 100;
};
rows.push({
name,
stars,
improvement,
ci95: ciPct(w95),
ci99: ciPct(w99),
ci999: ciPct(w999),
pValue: w95.pValue,
});
if (name.length > maxNameLen) maxNameLen = name.length;
}
// Print header.
const pad = (s, n) => s + ' '.repeat(Math.max(0, n - s.length));
const rpad = (s, n) => ' '.repeat(Math.max(0, n - s.length)) + s;
console.log(`${pad('', maxNameLen)} confidence` +
` improvement accuracy (*) (**) (***)`);
for (const row of rows) {
const imp = `${row.improvement >= 0 ? '+' : ''}${row.improvement.toFixed(2)} %`;
console.log(
`${pad(row.name, maxNameLen)} ${pad(row.stars, 10)}` +
` ${rpad(imp, 11)}` +
` ±${row.ci95.toFixed(2)}%` +
` ±${row.ci99.toFixed(2)}%` +
` ±${row.ci999.toFixed(2)}%`,
);
}
if (skipped > 0) {
console.log('');
console.log(
`Note: ${skipped} configuration${skipped === 1 ? ' was' : 's were'}` +
` skipped because Welch's t-test requires at least 2 samples per` +
` binary. Use --runs 2 or higher.`,
);
}
// --- Bar chart visualization ---
printChart(rows, maxNameLen);
console.log('');
console.log(
`Rates were scaled by ${scale}x into HdrHistogram (3 significant figures).\n` +
`Use --scale to adjust precision if needed.\n`,
);
console.log(
`Be aware that when doing many comparisons the risk of a false-positive\n` +
`result increases. In this case, there are ${rows.length} comparisons, ` +
`you can thus\nexpect the following amount of false-positive results:\n` +
` ${(rows.length * 0.05).toFixed(2)} false positives, when considering ` +
`a 5% risk acceptance (*, **, ***),\n` +
` ${(rows.length * 0.01).toFixed(2)} false positives, when considering ` +
`a 1% risk acceptance (**, ***),\n` +
` ${(rows.length * 0.001).toFixed(2)} false positives, when considering ` +
`a 0.1% risk acceptance (***)`,
);
// Gate: exit with error if any significant regression exceeds the limit.
if (maxRegression > 0) {
const failures = rows.filter(
(r) => r.stars.trim() !== '' && r.improvement < -maxRegression,
);
if (failures.length > 0) {
console.log('');
console.log(
`FAIL: ${failures.length} benchmark${failures.length === 1 ? '' : 's'}` +
` showed a statistically significant regression exceeding` +
` ${maxRegression}%:`,
);
for (const f of failures) {
console.log(` ${f.name} ${f.improvement.toFixed(2)}%`);
}
process.exitCode = 1;
}
}
}
function printChart(rows, maxNameLen) {
if (rows.length === 0) return;
// Determine the chart scale from the data. The bar region covers
// the range [-maxAbs, +maxAbs] so the zero line sits in the center.
const barWidth = 40;
const halfWidth = barWidth / 2;
let maxAbs = 0;
for (const row of rows) {
const extent = Math.abs(row.improvement) + row.ci95;
if (extent > maxAbs) maxAbs = extent;
}
if (maxAbs === 0) maxAbs = 1;
const pad = (s, n) => s + ' '.repeat(Math.max(0, n - s.length));
// Scale axis labels.
const axisLeft = `-${maxAbs.toFixed(1)}%`;
const axisRight = `+${maxAbs.toFixed(1)}%`;
const axisCenter = '0%';
// Print axis header.
const labelPad = maxNameLen + 5;
const leftLabel = ' '.repeat(labelPad) +
axisLeft +
' '.repeat(Math.max(0, halfWidth - axisLeft.length - Math.floor(axisCenter.length / 2))) +
axisCenter +
' '.repeat(Math.max(0, halfWidth - Math.ceil(axisCenter.length / 2) - axisRight.length)) +
axisRight;
console.log('');
console.log(leftLabel);
for (const row of rows) {
const imp = row.improvement;
const ci = row.ci95;
// Position of the improvement value in the bar region [0, barWidth].
const center = halfWidth;
const impPos = center + (imp / maxAbs) * halfWidth;
// CI extent in bar positions.
const ciLeft = center + ((imp - ci) / maxAbs) * halfWidth;
const ciRight = center + ((imp + ci) / maxAbs) * halfWidth;
// Build the bar character by character.
const chars = [];
for (let x = 0; x < barWidth; x++) {
const pos = x + 0.5; // Center of this character cell.
if (x === Math.floor(center)) {
chars.push('|');
} else if ((imp >= 0 && pos > center && pos <= impPos) ||
(imp < 0 && pos < center && pos >= impPos)) {
chars.push(row.stars ? '\u2588' : '\u2593'); // solid or dark shade
} else if (pos >= ciLeft && pos <= ciRight) {
chars.push('\u2591'); // Light shade for CI region
} else {
chars.push(' ');
}
}
const label = `${row.improvement >= 0 ? '+' : ''}${row.improvement.toFixed(2)}%`;
const sig = row.stars.trim();
console.log(`${pad(row.name, maxNameLen)} ${chars.join('')} ${label} ${sig}`);
}
}
@@ -14,6 +14,8 @@
* [Specifying CPU Cores for Benchmarks with run.js](#specifying-cpu-cores-for-benchmarks-with-runjs)
* [Filtering benchmarks](#filtering-benchmarks)
* [Comparing Node.js versions](#comparing-nodejs-versions)
* [Using `--analyze` (no external tools needed)](#using---analyze-no-external-tools-needed)
* [Using R scripts or node-benchmark-compare](#using-r-scripts-or-node-benchmark-compare)
* [Comparing parameters](#comparing-parameters)
* [Running benchmarks on the CI](#running-benchmarks-on-the-ci)
* [Creating a benchmark](#creating-a-benchmark)
@@ -73,18 +75,27 @@ node benchmark/http2/simple.js benchmarker=h2load
### Benchmark analysis requirements
To analyze the results statistically, you can use either the
[node-benchmark-compare][] tool or the R script `benchmark/compare.R`.
To analyze the results statistically, there are three options:
[node-benchmark-compare][] is a Node.js script that can be installed with
`npm install -g node-benchmark-compare`.
* **`--analyze` flag** (built-in, no dependencies): Pass `--analyze` to
`benchmark/compare.js` to perform Welch's t-test directly after the
benchmarks complete. This uses the histogram API's statistical testing
methods and requires no external tools.
* **R scripts** (`benchmark/compare.R`, `benchmark/bar.R`): Perform the same
Welch's t-test analysis as `--analyze`, with the additional ability to
generate plots. Requires R with the `ggplot2` and `plyr` packages.
* **[node-benchmark-compare][]** (legacy): A Node.js script that can be
installed with `npm install -g node-benchmark-compare`. It reads the CSV
output of `benchmark/compare.js`. Predates the built-in `--analyze` flag
and is no longer necessary for most workflows.
To draw comparison plots when analyzing the results, `R` must be installed.
Use one of the available package managers or download it from
<https://www.r-project.org/>.
For most use cases, `--analyze` is the simplest option since it requires
nothing beyond Node.js itself.
The R packages `ggplot2` and `plyr` are also used and can be installed using
the R REPL.
To install R for plot generation, use one of the available package managers or
download it from <https://www.r-project.org/>.
The R packages `ggplot2` and `plyr` can be installed using the R REPL.
```console
$ R
@@ -403,16 +414,38 @@ module, you can use the `--filter` option:_
repeated)
--set variable=value set benchmark variable (can be repeated)
--no-progress don't show benchmark progress indicator
Examples:
--set CPUSET=0 Runs benchmarks on CPU core 0.
--set CPUSET=0-2 Specifies that benchmarks should run on CPU cores 0 to 2.
Note: The CPUSET format should match the specifications of the 'taskset' command
--analyze perform statistical analysis inline (no R needed)
--scale 1000 rate multiplier for --analyze precision
--max-regression N exit with code 1 if any significant regression
exceeds N% (implies --analyze)
```
For analyzing the benchmark results, use [node-benchmark-compare][] or the R
scripts:
#### Using `--analyze` (no external tools needed)
The simplest way to get statistical results is to pass `--analyze`:
```bash
node benchmark/compare.js --old ./node-main --new ./node-pr-5134 --analyze string_decoder
```
This runs the benchmarks and prints the analysis directly:
```console
confidence improvement accuracy (*) (**) (***)
string_decoder/string-decoder.js n=2500000 chunkLen=16 inLen=128 encoding='ascii' *** -3.76 % ±1.36% ±1.82% ±2.40%
string_decoder/string-decoder.js n=2500000 chunkLen=16 inLen=128 encoding='utf8' ** -0.81 % ±0.53% ±0.71% ±0.93%
...
```
The `--analyze` mode uses the histogram API's `welchTest()` method to perform
the same Welch's t-test that the R script uses. Benchmark rates are scaled to
integers for the histogram (controlled by `--scale`, default 1000). With the
default settings, results are identical to the R script at two decimal places.
#### Using R scripts or node-benchmark-compare
Alternatively, save the CSV output and analyze it separately using
[node-benchmark-compare][] or the R scripts:
* `benchmark/compare.R`
* `benchmark/bar.R`
@@ -428,6 +461,10 @@ $ node-benchmark-compare compare-pr-5134.csv # or cat compare-pr-5134.csv | Rscr
...
```
The R approach is still useful when you need to generate plots (box plots via
`compare.R --plot`, scatter plots via `scatter.R --plot`) or when you want to
analyze previously saved CSV files.
In the output, _improvement_ is the relative improvement of the new version,
hopefully this is positive. _confidence_ tells if there is enough
statistical evidence to validate the _improvement_. If there is enough evidence