zlib: pledge input size in async zstdCompress()

zstdCompress() writes its input and ends the frame in separate calls,
so zstd cannot infer the input size the way it does for
zstdCompressSync(), and sizes its tables for an unbounded stream.

Default pledgedSrcSize to the input's byte length. The output is now
identical to zstdCompressSync() and several times faster at higher
levels. An explicit pledgedSrcSize, or a string with a custom
defaultEncoding, keeps the current behavior.

Signed-off-by: James Ross <james@jross.me>
PR-URL: https://github.com/nodejs/node/pull/66358
Reviewed-By: Vinícius Lourenço Claro Cardoso <contact@viniciusl.com.br>
Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
This commit is contained in:
James Ross authored and Node.js GitHub Bot committed 2026-09-29 21:23:49 +00:00
1 parent e6b3f96cd0
commit fa4af16414
3 files changed
+83 -2

No files matched your search

+9
View File
@@ -788,6 +788,9 @@ It's possible to specify the expected total size of the uncompressed input via
doesn't match at the end of the input, compression will fail with the code
`ZSTD_error_srcSize_wrong`.
[`zlib.zstdCompress()`][] defaults `opts.pledgedSrcSize` to the byte length of
its input.
#### Decompressor options
These advanced options are available for controlling decompression:
@@ -3072,6 +3075,11 @@ Decompress a chunk of data with [`Unzip`][].
added:
- v23.8.0
- v22.15.0
changes:
- version: REPLACEME
pr-url: https://github.com/nodejs/node/pull/66358
description: The `pledgedSrcSize` option defaults to the byte length of
`buffer`.
-->
* `buffer` {Buffer|TypedArray|DataView|ArrayBuffer|string}
@@ -3446,6 +3454,7 @@ Create a Zstandard decompression transform.
[`zlib.createZipArchive()`]: #zlibcreateziparchiveentries-options
[`zlib.createZipArchiveSync()`]: #zlibcreateziparchivesyncentries-options
[`zlib.getMaxZipContentSize()`]: #zlibgetmaxzipcontentsize
[`zlib.zstdCompress()`]: #zlibzstdcompressbuffer-options-callback
[convenience methods]: #convenience-methods
[zlib documentation]: https://zlib.net/manual.html#Constants
[zlib.createGzip example]: #zlib
+30 -2
View File
@@ -789,7 +789,7 @@ class Unzip extends Zlib {
}
}
function createConvenienceMethod(ctor, sync) {
function createConvenienceMethod(ctor, sync, prepareOpts) {
if (sync) {
return function syncBufferWrapper(buffer, opts) {
return zlibBufferSync(new ctor(opts), buffer);
@@ -800,10 +800,38 @@ function createConvenienceMethod(ctor, sync) {
callback = opts;
opts = {};
}
if (prepareOpts !== undefined) {
opts = prepareOpts(buffer, opts);
}
return zlibBuffer(new ctor(opts), buffer, callback);
};
}
// zstdCompress() writes the input and ends the frame in separate calls, so
// unlike zstdCompressSync() zstd cannot infer the input size and sizes its
// tables for an unbounded stream. Pledge the size, which is known up front.
function withPledgedSrcSize(buffer, opts) {
if (opts?.pledgedSrcSize !== undefined) {
return opts;
}
let pledgedSrcSize;
if (typeof buffer === 'string') {
// The stream encodes strings with defaultEncoding, so only a UTF-8 length
// is known to match what gets written.
const encoding = opts?.defaultEncoding;
if (encoding != null && encoding !== 'utf8' && encoding !== 'utf-8') {
return opts;
}
pledgedSrcSize = Buffer.byteLength(buffer);
} else if (isArrayBufferView(buffer) || isAnyArrayBuffer(buffer)) {
pledgedSrcSize = buffer.byteLength;
} else {
// Leave invalid input to the existing validation.
return opts;
}
return { __proto__: null, ...opts, pledgedSrcSize };
}
const kMaxBrotliParam = MathMax(
...ObjectEntries(constants)
.map(({ 0: key, 1: value }) => (key.startsWith('BROTLI_PARAM_') ? value : 0)),
@@ -1088,7 +1116,7 @@ module.exports = {
brotliCompressSync: createConvenienceMethod(BrotliCompress, true),
brotliDecompress: createConvenienceMethod(BrotliDecompress, false),
brotliDecompressSync: createConvenienceMethod(BrotliDecompress, true),
zstdCompress: createConvenienceMethod(ZstdCompress, false),
zstdCompress: createConvenienceMethod(ZstdCompress, false, withPledgedSrcSize),
zstdCompressSync: createConvenienceMethod(ZstdCompress, true),
zstdDecompress: createConvenienceMethod(ZstdDecompress, false),
zstdDecompressSync: createConvenienceMethod(ZstdDecompress, true),
@@ -112,3 +112,47 @@ for (const pledgedSrcSize of [
zlib.createZstdCompress({
pledgedSrcSize: Number.MAX_SAFE_INTEGER,
}).destroy();
// zstdCompress() pledges the input size by default, so its output matches
// zstdCompressSync(), which lets zstd infer the size from a single call.
{
const text = 'héllo wörld 🚀 '.repeat(1000);
const bytes = Buffer.from(text);
const inputs = [
'',
text,
bytes,
new Uint16Array(bytes.buffer, bytes.byteOffset, bytes.length >> 1),
new DataView(bytes.buffer, bytes.byteOffset, bytes.length),
bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.length),
];
const opts = {
params: { [zlib.constants.ZSTD_c_compressionLevel]: 9 },
};
for (const input of inputs) {
zlib.zstdCompress(input, opts, common.mustSucceed((compressed) => {
assert.deepStrictEqual(compressed, zlib.zstdCompressSync(input, opts));
}));
}
for (const defaultEncoding of ['utf8', 'utf-8']) {
const encodingOpts = { ...opts, defaultEncoding };
zlib.zstdCompress(text, encodingOpts, common.mustSucceed((compressed) => {
assert.deepStrictEqual(compressed, zlib.zstdCompressSync(text, encodingOpts));
}));
}
// The caller's options are left untouched, so they can be reused.
assert.strictEqual(opts.pledgedSrcSize, undefined);
// An explicit pledgedSrcSize is still honored.
zlib.zstdCompress(bytes, { pledgedSrcSize: 1 }, common.mustCall((err) => {
assert.strictEqual(err.code, pledgedSrcSizeError.code);
}));
// Strings written with a non-UTF-8 defaultEncoding still compress.
zlib.zstdCompress('é', { defaultEncoding: 'latin1' }, common.mustSucceed((compressed) => {
assert.deepStrictEqual(zlib.zstdDecompressSync(compressed), Buffer.from([0xe9]));
}));
}