Commit fa4af164144 for nodejs
commit fa4af164144545f15dbc46bc5fe72fef80da278a
Author: James Ross <james@jross.me>
Date: Sun Sep 27 22:14:49 2026 +0100
zlib: pledge input size in async zstdCompress()
zstdCompress() writes its input and ends the frame in separate calls,
so zstd cannot infer the input size the way it does for
zstdCompressSync(), and sizes its tables for an unbounded stream.
Default pledgedSrcSize to the input's byte length. The output is now
identical to zstdCompressSync() and several times faster at higher
levels. An explicit pledgedSrcSize, or a string with a custom
defaultEncoding, keeps the current behavior.
Signed-off-by: James Ross <james@jross.me>
PR-URL: https://github.com/nodejs/node/pull/66358
Reviewed-By: Vinícius Lourenço Claro Cardoso <contact@viniciusl.com.br>
Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
diff --git a/doc/api/zlib.md b/doc/api/zlib.md
index cef9e4b7937..50401f05776 100644
--- a/doc/api/zlib.md
+++ b/doc/api/zlib.md
@@ -788,6 +788,9 @@ It's possible to specify the expected total size of the uncompressed input via
doesn't match at the end of the input, compression will fail with the code
`ZSTD_error_srcSize_wrong`.
+[`zlib.zstdCompress()`][] defaults `opts.pledgedSrcSize` to the byte length of
+its input.
+
#### Decompressor options
These advanced options are available for controlling decompression:
@@ -3072,6 +3075,11 @@ Decompress a chunk of data with [`Unzip`][].
added:
- v23.8.0
- v22.15.0
+changes:
+ - version: REPLACEME
+ pr-url: https://github.com/nodejs/node/pull/66358
+ description: The `pledgedSrcSize` option defaults to the byte length of
+ `buffer`.
-->
* `buffer` {Buffer|TypedArray|DataView|ArrayBuffer|string}
@@ -3446,6 +3454,7 @@ Create a Zstandard decompression transform.
[`zlib.createZipArchive()`]: #zlibcreateziparchiveentries-options
[`zlib.createZipArchiveSync()`]: #zlibcreateziparchivesyncentries-options
[`zlib.getMaxZipContentSize()`]: #zlibgetmaxzipcontentsize
+[`zlib.zstdCompress()`]: #zlibzstdcompressbuffer-options-callback
[convenience methods]: #convenience-methods
[zlib documentation]: https://zlib.net/manual.html#Constants
[zlib.createGzip example]: #zlib
diff --git a/lib/zlib.js b/lib/zlib.js
index 4a2f488835e..f6c5b177b58 100644
--- a/lib/zlib.js
+++ b/lib/zlib.js
@@ -789,7 +789,7 @@ class Unzip extends Zlib {
}
}
-function createConvenienceMethod(ctor, sync) {
+function createConvenienceMethod(ctor, sync, prepareOpts) {
if (sync) {
return function syncBufferWrapper(buffer, opts) {
return zlibBufferSync(new ctor(opts), buffer);
@@ -800,10 +800,38 @@ function createConvenienceMethod(ctor, sync) {
callback = opts;
opts = {};
}
+ if (prepareOpts !== undefined) {
+ opts = prepareOpts(buffer, opts);
+ }
return zlibBuffer(new ctor(opts), buffer, callback);
};
}
+// zstdCompress() writes the input and ends the frame in separate calls, so
+// unlike zstdCompressSync() zstd cannot infer the input size and sizes its
+// tables for an unbounded stream. Pledge the size, which is known up front.
+function withPledgedSrcSize(buffer, opts) {
+ if (opts?.pledgedSrcSize !== undefined) {
+ return opts;
+ }
+ let pledgedSrcSize;
+ if (typeof buffer === 'string') {
+ // The stream encodes strings with defaultEncoding, so only a UTF-8 length
+ // is known to match what gets written.
+ const encoding = opts?.defaultEncoding;
+ if (encoding != null && encoding !== 'utf8' && encoding !== 'utf-8') {
+ return opts;
+ }
+ pledgedSrcSize = Buffer.byteLength(buffer);
+ } else if (isArrayBufferView(buffer) || isAnyArrayBuffer(buffer)) {
+ pledgedSrcSize = buffer.byteLength;
+ } else {
+ // Leave invalid input to the existing validation.
+ return opts;
+ }
+ return { __proto__: null, ...opts, pledgedSrcSize };
+}
+
const kMaxBrotliParam = MathMax(
...ObjectEntries(constants)
.map(({ 0: key, 1: value }) => (key.startsWith('BROTLI_PARAM_') ? value : 0)),
@@ -1088,7 +1116,7 @@ module.exports = {
brotliCompressSync: createConvenienceMethod(BrotliCompress, true),
brotliDecompress: createConvenienceMethod(BrotliDecompress, false),
brotliDecompressSync: createConvenienceMethod(BrotliDecompress, true),
- zstdCompress: createConvenienceMethod(ZstdCompress, false),
+ zstdCompress: createConvenienceMethod(ZstdCompress, false, withPledgedSrcSize),
zstdCompressSync: createConvenienceMethod(ZstdCompress, true),
zstdDecompress: createConvenienceMethod(ZstdDecompress, false),
zstdDecompressSync: createConvenienceMethod(ZstdDecompress, true),
diff --git a/test/parallel/test-zlib-zstd-pledged-src-size.js b/test/parallel/test-zlib-zstd-pledged-src-size.js
index a0b1babd3f7..d30a4aadefd 100644
--- a/test/parallel/test-zlib-zstd-pledged-src-size.js
+++ b/test/parallel/test-zlib-zstd-pledged-src-size.js
@@ -112,3 +112,47 @@ for (const pledgedSrcSize of [
zlib.createZstdCompress({
pledgedSrcSize: Number.MAX_SAFE_INTEGER,
}).destroy();
+
+// zstdCompress() pledges the input size by default, so its output matches
+// zstdCompressSync(), which lets zstd infer the size from a single call.
+{
+ const text = 'héllo wörld 🚀 '.repeat(1000);
+ const bytes = Buffer.from(text);
+ const inputs = [
+ '',
+ text,
+ bytes,
+ new Uint16Array(bytes.buffer, bytes.byteOffset, bytes.length >> 1),
+ new DataView(bytes.buffer, bytes.byteOffset, bytes.length),
+ bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.length),
+ ];
+ const opts = {
+ params: { [zlib.constants.ZSTD_c_compressionLevel]: 9 },
+ };
+
+ for (const input of inputs) {
+ zlib.zstdCompress(input, opts, common.mustSucceed((compressed) => {
+ assert.deepStrictEqual(compressed, zlib.zstdCompressSync(input, opts));
+ }));
+ }
+
+ for (const defaultEncoding of ['utf8', 'utf-8']) {
+ const encodingOpts = { ...opts, defaultEncoding };
+ zlib.zstdCompress(text, encodingOpts, common.mustSucceed((compressed) => {
+ assert.deepStrictEqual(compressed, zlib.zstdCompressSync(text, encodingOpts));
+ }));
+ }
+
+ // The caller's options are left untouched, so they can be reused.
+ assert.strictEqual(opts.pledgedSrcSize, undefined);
+
+ // An explicit pledgedSrcSize is still honored.
+ zlib.zstdCompress(bytes, { pledgedSrcSize: 1 }, common.mustCall((err) => {
+ assert.strictEqual(err.code, pledgedSrcSizeError.code);
+ }));
+
+ // Strings written with a non-UTF-8 defaultEncoding still compress.
+ zlib.zstdCompress('é', { defaultEncoding: 'latin1' }, common.mustSucceed((compressed) => {
+ assert.deepStrictEqual(zlib.zstdDecompressSync(compressed), Buffer.from([0xe9]));
+ }));
+}