Skip to content

Commit 0b56e6f

Browse files
committed
zlib: pledge input size in async zstdCompress()
zstdCompress() writes its input and ends the frame in separate calls, so zstd cannot infer the input size the way it does for zstdCompressSync(), and sizes its tables for an unbounded stream. Default pledgedSrcSize to the input's byte length. The output is now identical to zstdCompressSync() and several times faster at higher levels. An explicit pledgedSrcSize, or a string with a custom defaultEncoding, keeps the current behavior. Signed-off-by: James Ross <james@jross.me>
1 parent a2a064c commit 0b56e6f

3 files changed

Lines changed: 75 additions & 2 deletions

File tree

‎doc/api/zlib.md‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -788,6 +788,9 @@ It's possible to specify the expected total size of the uncompressed input via
788788
doesn't match at the end of the input, compression will fail with the code
789789
`ZSTD_error_srcSize_wrong`.
790790

791+
[`zlib.zstdCompress()`][] defaults `opts.pledgedSrcSize` to the byte length of
792+
its input.
793+
791794
#### Decompressor options
792795

793796
These advanced options are available for controlling decompression:
@@ -3072,6 +3075,11 @@ Decompress a chunk of data with [`Unzip`][].
30723075
added:
30733076
- v23.8.0
30743077
- v22.15.0
3078+
changes:
3079+
- version: REPLACEME
3080+
pr-url: https://github.com/nodejs/node/pull/XXXXX
3081+
description: The `pledgedSrcSize` option defaults to the byte length of
3082+
`buffer`.
30753083
-->
30763084

30773085
* `buffer` {Buffer|TypedArray|DataView|ArrayBuffer|string}
@@ -3446,6 +3454,7 @@ Create a Zstandard decompression transform.
34463454
[`zlib.createZipArchive()`]: #zlibcreateziparchiveentries-options
34473455
[`zlib.createZipArchiveSync()`]: #zlibcreateziparchivesyncentries-options
34483456
[`zlib.getMaxZipContentSize()`]: #zlibgetmaxzipcontentsize
3457+
[`zlib.zstdCompress()`]: #zlibzstdcompressbuffer-options-callback
34493458
[convenience methods]: #convenience-methods
34503459
[zlib documentation]: https://zlib.net/manual.html#Constants
34513460
[zlib.createGzip example]: #zlib

‎lib/zlib.js‎

Lines changed: 29 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -789,7 +789,7 @@ class Unzip extends Zlib {
789789
}
790790
}
791791

792-
function createConvenienceMethod(ctor, sync) {
792+
function createConvenienceMethod(ctor, sync, prepareOpts) {
793793
if (sync) {
794794
return function syncBufferWrapper(buffer, opts) {
795795
return zlibBufferSync(new ctor(opts), buffer);
@@ -800,10 +800,37 @@ function createConvenienceMethod(ctor, sync) {
800800
callback = opts;
801801
opts = {};
802802
}
803+
if (prepareOpts !== undefined) {
804+
opts = prepareOpts(buffer, opts);
805+
}
803806
return zlibBuffer(new ctor(opts), buffer, callback);
804807
};
805808
}
806809

810+
// zstdCompress() writes the input and ends the frame in separate calls, so
811+
// unlike zstdCompressSync() zstd cannot infer the input size and sizes its
812+
// tables for an unbounded stream. Pledge the size, which is known up front.
813+
function withPledgedSrcSize(buffer, opts) {
814+
if (opts?.pledgedSrcSize !== undefined) {
815+
return opts;
816+
}
817+
let pledgedSrcSize;
818+
if (typeof buffer === 'string') {
819+
// The stream encodes strings with defaultEncoding, so only a UTF-8 length
820+
// is known to match what gets written.
821+
if (opts?.defaultEncoding !== undefined) {
822+
return opts;
823+
}
824+
pledgedSrcSize = Buffer.byteLength(buffer);
825+
} else if (isArrayBufferView(buffer) || isAnyArrayBuffer(buffer)) {
826+
pledgedSrcSize = buffer.byteLength;
827+
} else {
828+
// Leave invalid input to the existing validation.
829+
return opts;
830+
}
831+
return { __proto__: null, ...opts, pledgedSrcSize };
832+
}
833+
807834
const kMaxBrotliParam = MathMax(
808835
...ObjectEntries(constants)
809836
.map(({ 0: key, 1: value }) => (key.startsWith('BROTLI_PARAM_') ? value : 0)),
@@ -1088,7 +1115,7 @@ module.exports = {
10881115
brotliCompressSync: createConvenienceMethod(BrotliCompress, true),
10891116
brotliDecompress: createConvenienceMethod(BrotliDecompress, false),
10901117
brotliDecompressSync: createConvenienceMethod(BrotliDecompress, true),
1091-
zstdCompress: createConvenienceMethod(ZstdCompress, false),
1118+
zstdCompress: createConvenienceMethod(ZstdCompress, false, withPledgedSrcSize),
10921119
zstdCompressSync: createConvenienceMethod(ZstdCompress, true),
10931120
zstdDecompress: createConvenienceMethod(ZstdDecompress, false),
10941121
zstdDecompressSync: createConvenienceMethod(ZstdDecompress, true),

‎test/parallel/test-zlib-zstd-pledged-src-size.js‎

Lines changed: 37 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -112,3 +112,40 @@ for (const pledgedSrcSize of [
112112
zlib.createZstdCompress({
113113
pledgedSrcSize: Number.MAX_SAFE_INTEGER,
114114
}).destroy();
115+
116+
// zstdCompress() pledges the input size by default, so its output matches
117+
// zstdCompressSync(), which lets zstd infer the size from a single call.
118+
{
119+
const text = 'héllo wörld 🚀 '.repeat(1000);
120+
const bytes = Buffer.from(text);
121+
const inputs = [
122+
'',
123+
text,
124+
bytes,
125+
new Uint16Array(bytes.buffer, bytes.byteOffset, bytes.length >> 1),
126+
new DataView(bytes.buffer, bytes.byteOffset, bytes.length),
127+
bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.length),
128+
];
129+
const opts = {
130+
params: { [zlib.constants.ZSTD_c_compressionLevel]: 9 },
131+
};
132+
133+
for (const input of inputs) {
134+
zlib.zstdCompress(input, opts, common.mustSucceed((compressed) => {
135+
assert.deepStrictEqual(compressed, zlib.zstdCompressSync(input, opts));
136+
}));
137+
}
138+
139+
// The caller's options are left untouched, so they can be reused.
140+
assert.strictEqual(opts.pledgedSrcSize, undefined);
141+
142+
// An explicit pledgedSrcSize is still honored.
143+
zlib.zstdCompress(bytes, { pledgedSrcSize: 1 }, common.mustCall((err) => {
144+
assert.strictEqual(err.code, pledgedSrcSizeError.code);
145+
}));
146+
147+
// Strings written with a non-UTF-8 defaultEncoding still compress.
148+
zlib.zstdCompress('é', { defaultEncoding: 'latin1' }, common.mustSucceed((compressed) => {
149+
assert.deepStrictEqual(zlib.zstdDecompressSync(compressed), Buffer.from([0xe9]));
150+
}));
151+
}

0 commit comments

Comments
 (0)