mirror of
https://github.com/nodejs/node.git
synced 2026-10-10 18:59:49 -04:00
benchmark: add --analyze mode to compare.js
Add an --analyze flag that performs statistical analysis directly
after benchmarks complete, eliminating the need for R and compare.R.
When --analyze is specified, compare.js collects the rate data during
the run and prints a statistical summary table instead of CSV output.
The table matches the format of compare.R: improvement percentage,
significance stars (* p<0.05, ** p<0.01, *** p<0.001), and confidence
intervals at three risk levels.
Also adds a --max-regression N option that causes the compare.js to
exit with 1 (error) when the `--new` is N% slower. Useful for CI
use to detect regressions.
Uses the histogram API's welchTest() and cohensD() methods introduced
in the previous commit. Benchmark rates are scaled to integers for
HdrHistogram recording; the --scale option (default 1000) controls
the multiplier for precision.
Usage:
node benchmark/compare.js --old ./node-old --new ./node-new \
--analyze url
Signed-off-by: James M Snell <jasnell@gmail.com>
Assisted-by: Opencode/Opus
PR-URL: https://github.com/nodejs/node/pull/65416
Reviewed-By: Matteo Collina <matteo.collina@gmail.com>
Reviewed-By: Chengzhong Wu <legendecas@gmail.com>
This commit is contained in:
1 parent
524dee4372
commit
bf67fdc6d7
3 files changed
+290
-28
No files matched your search
@@ -25,9 +25,10 @@ function getTime(diff) {
|
||||
// A run is an item in the job queue: { binary, filename, iter }
|
||||
// A config is an item in the subqueue: { binary, filename, iter, configs }
|
||||
class BenchmarkProgress {
|
||||
constructor(queue, benchmarks) {
|
||||
constructor(queue, benchmarks, options = {}) {
|
||||
this.queue = queue; // Scheduled runs.
|
||||
this.benchmarks = benchmarks; // Filenames of scheduled benchmarks.
|
||||
this.analyze = !!options.analyze; // stdout is not piped, but unused.
|
||||
this.completedRuns = 0; // Number of completed runs.
|
||||
this.scheduledRuns = queue.length; // Number of scheduled runs.
|
||||
// Time when starting to run benchmarks.
|
||||
@@ -107,7 +108,10 @@ class BenchmarkProgress {
|
||||
}
|
||||
|
||||
updateProgress() {
|
||||
if (!process.stderr.isTTY || process.stdout.isTTY) {
|
||||
// Progress renders on stderr when stdout is piped (not a TTY).
|
||||
// In --analyze mode, stdout is the terminal but is unused during
|
||||
// the run, so treat it the same as piped.
|
||||
if (!process.stderr.isTTY || (process.stdout.isTTY && !this.analyze)) {
|
||||
return;
|
||||
}
|
||||
readline.clearLine(process.stderr);
|
||||
|
||||
+230
-9
@@ -13,7 +13,8 @@ const cli = new CLI(`usage: ./node compare.js [options] [--] <category> ...
|
||||
Run each benchmark in the <category> directory many times using two different
|
||||
node versions. More than one <category> directory can be specified.
|
||||
The output is formatted as csv, which can be processed using for
|
||||
example 'compare.R'.
|
||||
example 'compare.R'. Use --analyze to perform statistical analysis
|
||||
directly without R.
|
||||
|
||||
--new ./new-node-binary new node binary (required)
|
||||
--old ./old-node-binary old node binary (required)
|
||||
@@ -24,13 +25,21 @@ const cli = new CLI(`usage: ./node compare.js [options] [--] <category> ...
|
||||
repeated)
|
||||
--set variable=value set benchmark variable (can be repeated)
|
||||
--no-progress don't show benchmark progress indicator
|
||||
--analyze perform statistical analysis after benchmarks
|
||||
complete (Welch's t-test, effect size) instead
|
||||
of printing csv output
|
||||
--scale 1000 rate-to-integer multiplier for histogram
|
||||
precision when using --analyze (default: 1000)
|
||||
--max-regression N exit with code 1 if any statistically
|
||||
significant regression exceeds N% (implies
|
||||
--analyze)
|
||||
|
||||
Examples:
|
||||
--set CPUSET=0 Runs benchmarks on CPU core 0.
|
||||
--set CPUSET=0-2 Specifies that benchmarks should run on CPU cores 0 to 2.
|
||||
|
||||
Note: The CPUSET format should match the specifications of the 'taskset' command
|
||||
`, { arrayArgs: ['set', 'filter', 'exclude'], boolArgs: ['no-progress'] });
|
||||
`, { arrayArgs: ['set', 'filter', 'exclude'], boolArgs: ['no-progress', 'analyze'] });
|
||||
|
||||
if (!cli.optional.new || !cli.optional.old) {
|
||||
cli.abort(cli.usage);
|
||||
@@ -38,6 +47,11 @@ if (!cli.optional.new || !cli.optional.old) {
|
||||
|
||||
const binaries = ['old', 'new'];
|
||||
const runs = cli.optional.runs ? parseInt(cli.optional.runs, 10) : 30;
|
||||
const maxRegression = cli.optional['max-regression'] ?
|
||||
parseFloat(cli.optional['max-regression']) :
|
||||
0;
|
||||
const analyze = !!cli.optional.analyze || maxRegression > 0;
|
||||
const scale = cli.optional.scale ? parseInt(cli.optional.scale, 10) : 1000;
|
||||
const benchmarks = cli.benchmarks();
|
||||
|
||||
if (benchmarks.length === 0) {
|
||||
@@ -46,6 +60,9 @@ if (benchmarks.length === 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// When --analyze is set, collect results for statistical analysis.
|
||||
const results = analyze ? new Map() : null;
|
||||
|
||||
// Create queue from the benchmarks list such both node versions are tested
|
||||
// `runs` amount of times each.
|
||||
// Note: BenchmarkProgress relies on this order to estimate
|
||||
@@ -61,15 +78,17 @@ for (const filename of benchmarks) {
|
||||
}
|
||||
// queue.length = binary.length * runs * benchmarks.length
|
||||
|
||||
// Print csv header
|
||||
console.log('"binary","filename","configuration","rate","time"');
|
||||
// Print csv header (unless analyzing inline).
|
||||
if (!analyze) {
|
||||
console.log('"binary","filename","configuration","rate","time"');
|
||||
}
|
||||
|
||||
const kStartOfQueue = 0;
|
||||
|
||||
const showProgress = !cli.optional['no-progress'];
|
||||
let progress;
|
||||
if (showProgress) {
|
||||
progress = new BenchmarkProgress(queue, benchmarks);
|
||||
progress = new BenchmarkProgress(queue, benchmarks, { analyze });
|
||||
progress.startQueue(kStartOfQueue);
|
||||
}
|
||||
|
||||
@@ -99,11 +118,20 @@ if (showProgress) {
|
||||
conf += ` ${key}=${inspect(data.conf[key])}`;
|
||||
}
|
||||
conf = conf.slice(1);
|
||||
// Escape quotes (") for correct csv formatting
|
||||
conf = conf.replace(/"/g, '""');
|
||||
|
||||
console.log(`"${job.binary}","${job.filename}","${conf}",` +
|
||||
`${data.rate},${data.time}`);
|
||||
if (analyze) {
|
||||
// Collect results for post-run analysis.
|
||||
const name = `${job.filename} ${conf}`;
|
||||
if (!results.has(name)) {
|
||||
results.set(name, { old: [], new: [] });
|
||||
}
|
||||
results.get(name)[job.binary].push(data.rate);
|
||||
} else {
|
||||
// Escape quotes (") for correct csv formatting
|
||||
conf = conf.replace(/"/g, '""');
|
||||
console.log(`"${job.binary}","${job.filename}","${conf}",` +
|
||||
`${data.rate},${data.time}`);
|
||||
}
|
||||
if (showProgress) {
|
||||
// One item in the subqueue has been completed.
|
||||
progress.completeConfig(data);
|
||||
@@ -125,6 +153,199 @@ if (showProgress) {
|
||||
// If there are more benchmarks execute the next
|
||||
if (i + 1 < queue.length) {
|
||||
recursive(i + 1);
|
||||
} else if (analyze) {
|
||||
printAnalysis(results, scale, maxRegression);
|
||||
}
|
||||
});
|
||||
})(kStartOfQueue);
|
||||
|
||||
function printAnalysis(results, scale, maxRegression) {
|
||||
const { createHistogram } = require('node:perf_hooks');
|
||||
|
||||
// Build per-benchmark histograms and run statistical tests.
|
||||
const rows = [];
|
||||
let maxNameLen = 0;
|
||||
|
||||
let skipped = 0;
|
||||
|
||||
for (const [name, { old: oldRates, new: newRates }] of results) {
|
||||
if (oldRates.length < 2 || newRates.length < 2) {
|
||||
skipped++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const hOld = createHistogram({ figures: 3 });
|
||||
const hNew = createHistogram({ figures: 3 });
|
||||
|
||||
for (const r of oldRates) hOld.record(Math.max(1, Math.round(r * scale)));
|
||||
for (const r of newRates) hNew.record(Math.max(1, Math.round(r * scale)));
|
||||
|
||||
const oldMean = oldRates.reduce((a, b) => a + b, 0) / oldRates.length;
|
||||
const newMean = newRates.reduce((a, b) => a + b, 0) / newRates.length;
|
||||
const improvement = ((newMean - oldMean) / oldMean) * 100;
|
||||
|
||||
// Query the three confidence levels. The p-value and t-statistic
|
||||
// are the same regardless of the confidence level, so we extract
|
||||
// them from the first result.
|
||||
const w95 = hOld.welchTest(hNew, { confidence: 0.95 });
|
||||
const w99 = hOld.welchTest(hNew, { confidence: 0.99 });
|
||||
const w999 = hOld.welchTest(hNew, { confidence: 0.999 });
|
||||
|
||||
// Significance stars matching compare.R convention.
|
||||
let stars = '';
|
||||
if (w95.pValue < 0.001) stars = '***';
|
||||
else if (w95.pValue < 0.01) stars = ' **';
|
||||
else if (w95.pValue < 0.05) stars = ' *';
|
||||
|
||||
// Confidence intervals expressed as percentage of the old mean.
|
||||
const ciPct = (w) => {
|
||||
const half =
|
||||
(w.confidenceInterval.upper - w.confidenceInterval.lower) / 2;
|
||||
return (half / (oldMean * scale)) * 100;
|
||||
};
|
||||
|
||||
rows.push({
|
||||
name,
|
||||
stars,
|
||||
improvement,
|
||||
ci95: ciPct(w95),
|
||||
ci99: ciPct(w99),
|
||||
ci999: ciPct(w999),
|
||||
pValue: w95.pValue,
|
||||
});
|
||||
|
||||
if (name.length > maxNameLen) maxNameLen = name.length;
|
||||
}
|
||||
|
||||
// Print header.
|
||||
const pad = (s, n) => s + ' '.repeat(Math.max(0, n - s.length));
|
||||
const rpad = (s, n) => ' '.repeat(Math.max(0, n - s.length)) + s;
|
||||
|
||||
console.log(`${pad('', maxNameLen)} confidence` +
|
||||
` improvement accuracy (*) (**) (***)`);
|
||||
|
||||
for (const row of rows) {
|
||||
const imp = `${row.improvement >= 0 ? '+' : ''}${row.improvement.toFixed(2)} %`;
|
||||
console.log(
|
||||
`${pad(row.name, maxNameLen)} ${pad(row.stars, 10)}` +
|
||||
` ${rpad(imp, 11)}` +
|
||||
` ±${row.ci95.toFixed(2)}%` +
|
||||
` ±${row.ci99.toFixed(2)}%` +
|
||||
` ±${row.ci999.toFixed(2)}%`,
|
||||
);
|
||||
}
|
||||
|
||||
if (skipped > 0) {
|
||||
console.log('');
|
||||
console.log(
|
||||
`Note: ${skipped} configuration${skipped === 1 ? ' was' : 's were'}` +
|
||||
` skipped because Welch's t-test requires at least 2 samples per` +
|
||||
` binary. Use --runs 2 or higher.`,
|
||||
);
|
||||
}
|
||||
|
||||
// --- Bar chart visualization ---
|
||||
printChart(rows, maxNameLen);
|
||||
|
||||
console.log('');
|
||||
console.log(
|
||||
`Rates were scaled by ${scale}x into HdrHistogram (3 significant figures).\n` +
|
||||
`Use --scale to adjust precision if needed.\n`,
|
||||
);
|
||||
console.log(
|
||||
`Be aware that when doing many comparisons the risk of a false-positive\n` +
|
||||
`result increases. In this case, there are ${rows.length} comparisons, ` +
|
||||
`you can thus\nexpect the following amount of false-positive results:\n` +
|
||||
` ${(rows.length * 0.05).toFixed(2)} false positives, when considering ` +
|
||||
`a 5% risk acceptance (*, **, ***),\n` +
|
||||
` ${(rows.length * 0.01).toFixed(2)} false positives, when considering ` +
|
||||
`a 1% risk acceptance (**, ***),\n` +
|
||||
` ${(rows.length * 0.001).toFixed(2)} false positives, when considering ` +
|
||||
`a 0.1% risk acceptance (***)`,
|
||||
);
|
||||
|
||||
// Gate: exit with error if any significant regression exceeds the limit.
|
||||
if (maxRegression > 0) {
|
||||
const failures = rows.filter(
|
||||
(r) => r.stars.trim() !== '' && r.improvement < -maxRegression,
|
||||
);
|
||||
if (failures.length > 0) {
|
||||
console.log('');
|
||||
console.log(
|
||||
`FAIL: ${failures.length} benchmark${failures.length === 1 ? '' : 's'}` +
|
||||
` showed a statistically significant regression exceeding` +
|
||||
` ${maxRegression}%:`,
|
||||
);
|
||||
for (const f of failures) {
|
||||
console.log(` ${f.name} ${f.improvement.toFixed(2)}%`);
|
||||
}
|
||||
process.exitCode = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function printChart(rows, maxNameLen) {
|
||||
if (rows.length === 0) return;
|
||||
|
||||
// Determine the chart scale from the data. The bar region covers
|
||||
// the range [-maxAbs, +maxAbs] so the zero line sits in the center.
|
||||
const barWidth = 40;
|
||||
const halfWidth = barWidth / 2;
|
||||
let maxAbs = 0;
|
||||
for (const row of rows) {
|
||||
const extent = Math.abs(row.improvement) + row.ci95;
|
||||
if (extent > maxAbs) maxAbs = extent;
|
||||
}
|
||||
if (maxAbs === 0) maxAbs = 1;
|
||||
|
||||
const pad = (s, n) => s + ' '.repeat(Math.max(0, n - s.length));
|
||||
|
||||
// Scale axis labels.
|
||||
const axisLeft = `-${maxAbs.toFixed(1)}%`;
|
||||
const axisRight = `+${maxAbs.toFixed(1)}%`;
|
||||
const axisCenter = '0%';
|
||||
|
||||
// Print axis header.
|
||||
const labelPad = maxNameLen + 5;
|
||||
const leftLabel = ' '.repeat(labelPad) +
|
||||
axisLeft +
|
||||
' '.repeat(Math.max(0, halfWidth - axisLeft.length - Math.floor(axisCenter.length / 2))) +
|
||||
axisCenter +
|
||||
' '.repeat(Math.max(0, halfWidth - Math.ceil(axisCenter.length / 2) - axisRight.length)) +
|
||||
axisRight;
|
||||
console.log('');
|
||||
console.log(leftLabel);
|
||||
|
||||
for (const row of rows) {
|
||||
const imp = row.improvement;
|
||||
const ci = row.ci95;
|
||||
|
||||
// Position of the improvement value in the bar region [0, barWidth].
|
||||
const center = halfWidth;
|
||||
const impPos = center + (imp / maxAbs) * halfWidth;
|
||||
|
||||
// CI extent in bar positions.
|
||||
const ciLeft = center + ((imp - ci) / maxAbs) * halfWidth;
|
||||
const ciRight = center + ((imp + ci) / maxAbs) * halfWidth;
|
||||
|
||||
// Build the bar character by character.
|
||||
const chars = [];
|
||||
for (let x = 0; x < barWidth; x++) {
|
||||
const pos = x + 0.5; // Center of this character cell.
|
||||
if (x === Math.floor(center)) {
|
||||
chars.push('|');
|
||||
} else if ((imp >= 0 && pos > center && pos <= impPos) ||
|
||||
(imp < 0 && pos < center && pos >= impPos)) {
|
||||
chars.push(row.stars ? '\u2588' : '\u2593'); // solid or dark shade
|
||||
} else if (pos >= ciLeft && pos <= ciRight) {
|
||||
chars.push('\u2591'); // Light shade for CI region
|
||||
} else {
|
||||
chars.push(' ');
|
||||
}
|
||||
}
|
||||
|
||||
const label = `${row.improvement >= 0 ? '+' : ''}${row.improvement.toFixed(2)}%`;
|
||||
const sig = row.stars.trim();
|
||||
console.log(`${pad(row.name, maxNameLen)} ${chars.join('')} ${label} ${sig}`);
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,8 @@
|
||||
* [Specifying CPU Cores for Benchmarks with run.js](#specifying-cpu-cores-for-benchmarks-with-runjs)
|
||||
* [Filtering benchmarks](#filtering-benchmarks)
|
||||
* [Comparing Node.js versions](#comparing-nodejs-versions)
|
||||
* [Using `--analyze` (no external tools needed)](#using---analyze-no-external-tools-needed)
|
||||
* [Using R scripts or node-benchmark-compare](#using-r-scripts-or-node-benchmark-compare)
|
||||
* [Comparing parameters](#comparing-parameters)
|
||||
* [Running benchmarks on the CI](#running-benchmarks-on-the-ci)
|
||||
* [Creating a benchmark](#creating-a-benchmark)
|
||||
@@ -73,18 +75,27 @@ node benchmark/http2/simple.js benchmarker=h2load
|
||||
|
||||
### Benchmark analysis requirements
|
||||
|
||||
To analyze the results statistically, you can use either the
|
||||
[node-benchmark-compare][] tool or the R script `benchmark/compare.R`.
|
||||
To analyze the results statistically, there are three options:
|
||||
|
||||
[node-benchmark-compare][] is a Node.js script that can be installed with
|
||||
`npm install -g node-benchmark-compare`.
|
||||
* **`--analyze` flag** (built-in, no dependencies): Pass `--analyze` to
|
||||
`benchmark/compare.js` to perform Welch's t-test directly after the
|
||||
benchmarks complete. This uses the histogram API's statistical testing
|
||||
methods and requires no external tools.
|
||||
* **R scripts** (`benchmark/compare.R`, `benchmark/bar.R`): Perform the same
|
||||
Welch's t-test analysis as `--analyze`, with the additional ability to
|
||||
generate plots. Requires R with the `ggplot2` and `plyr` packages.
|
||||
* **[node-benchmark-compare][]** (legacy): A Node.js script that can be
|
||||
installed with `npm install -g node-benchmark-compare`. It reads the CSV
|
||||
output of `benchmark/compare.js`. Predates the built-in `--analyze` flag
|
||||
and is no longer necessary for most workflows.
|
||||
|
||||
To draw comparison plots when analyzing the results, `R` must be installed.
|
||||
Use one of the available package managers or download it from
|
||||
<https://www.r-project.org/>.
|
||||
For most use cases, `--analyze` is the simplest option since it requires
|
||||
nothing beyond Node.js itself.
|
||||
|
||||
The R packages `ggplot2` and `plyr` are also used and can be installed using
|
||||
the R REPL.
|
||||
To install R for plot generation, use one of the available package managers or
|
||||
download it from <https://www.r-project.org/>.
|
||||
|
||||
The R packages `ggplot2` and `plyr` can be installed using the R REPL.
|
||||
|
||||
```console
|
||||
$ R
|
||||
@@ -403,16 +414,38 @@ module, you can use the `--filter` option:_
|
||||
repeated)
|
||||
--set variable=value set benchmark variable (can be repeated)
|
||||
--no-progress don't show benchmark progress indicator
|
||||
|
||||
Examples:
|
||||
--set CPUSET=0 Runs benchmarks on CPU core 0.
|
||||
--set CPUSET=0-2 Specifies that benchmarks should run on CPU cores 0 to 2.
|
||||
|
||||
Note: The CPUSET format should match the specifications of the 'taskset' command
|
||||
--analyze perform statistical analysis inline (no R needed)
|
||||
--scale 1000 rate multiplier for --analyze precision
|
||||
--max-regression N exit with code 1 if any significant regression
|
||||
exceeds N% (implies --analyze)
|
||||
```
|
||||
|
||||
For analyzing the benchmark results, use [node-benchmark-compare][] or the R
|
||||
scripts:
|
||||
#### Using `--analyze` (no external tools needed)
|
||||
|
||||
The simplest way to get statistical results is to pass `--analyze`:
|
||||
|
||||
```bash
|
||||
node benchmark/compare.js --old ./node-main --new ./node-pr-5134 --analyze string_decoder
|
||||
```
|
||||
|
||||
This runs the benchmarks and prints the analysis directly:
|
||||
|
||||
```console
|
||||
confidence improvement accuracy (*) (**) (***)
|
||||
string_decoder/string-decoder.js n=2500000 chunkLen=16 inLen=128 encoding='ascii' *** -3.76 % ±1.36% ±1.82% ±2.40%
|
||||
string_decoder/string-decoder.js n=2500000 chunkLen=16 inLen=128 encoding='utf8' ** -0.81 % ±0.53% ±0.71% ±0.93%
|
||||
...
|
||||
```
|
||||
|
||||
The `--analyze` mode uses the histogram API's `welchTest()` method to perform
|
||||
the same Welch's t-test that the R script uses. Benchmark rates are scaled to
|
||||
integers for the histogram (controlled by `--scale`, default 1000). With the
|
||||
default settings, results are identical to the R script at two decimal places.
|
||||
|
||||
#### Using R scripts or node-benchmark-compare
|
||||
|
||||
Alternatively, save the CSV output and analyze it separately using
|
||||
[node-benchmark-compare][] or the R scripts:
|
||||
|
||||
* `benchmark/compare.R`
|
||||
* `benchmark/bar.R`
|
||||
@@ -428,6 +461,10 @@ $ node-benchmark-compare compare-pr-5134.csv # or cat compare-pr-5134.csv | Rscr
|
||||
...
|
||||
```
|
||||
|
||||
The R approach is still useful when you need to generate plots (box plots via
|
||||
`compare.R --plot`, scatter plots via `scatter.R --plot`) or when you want to
|
||||
analyze previously saved CSV files.
|
||||
|
||||
In the output, _improvement_ is the relative improvement of the new version,
|
||||
hopefully this is positive. _confidence_ tells if there is enough
|
||||
statistical evidence to validate the _improvement_. If there is enough evidence
|
||||
|
||||
Reference in new issue
Block a user