Skip to content

Commit 9b350d8

Browse files
Cherryaduh95
authored andcommitted
zlib: pledge input size in async zstdCompress()
zstdCompress() writes its input and ends the frame in separate calls, so zstd cannot infer the input size the way it does for zstdCompressSync(), and sizes its tables for an unbounded stream. Default pledgedSrcSize to the input's byte length. The output is now identical to zstdCompressSync() and several times faster at higher levels. An explicit pledgedSrcSize, or a string with a custom defaultEncoding, keeps the current behavior. Signed-off-by: James Ross <james@jross.me> PR-URL: #66358 Reviewed-By: Vinícius Lourenço Claro Cardoso <contact@viniciusl.com.br> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent 6c23e21 commit 9b350d8

3 files changed

Lines changed: 83 additions & 2 deletions

File tree

‎doc/api/zlib.md‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -788,6 +788,9 @@ It's possible to specify the expected total size of the uncompressed input via
788788
doesn't match at the end of the input, compression will fail with the code
789789
`ZSTD_error_srcSize_wrong`.
790790

791+
[`zlib.zstdCompress()`][] defaults `opts.pledgedSrcSize` to the byte length of
792+
its input.
793+
791794
#### Decompressor options
792795

793796
These advanced options are available for controlling decompression:
@@ -3064,6 +3067,11 @@ Decompress a chunk of data with [`Unzip`][].
30643067
added:
30653068
- v23.8.0
30663069
- v22.15.0
3070+
changes:
3071+
- version: REPLACEME
3072+
pr-url: https://github.com/nodejs/node/pull/66358
3073+
description: The `pledgedSrcSize` option defaults to the byte length of
3074+
`buffer`.
30673075
-->
30683076

30693077
* `buffer` {Buffer|TypedArray|DataView|ArrayBuffer|string}
@@ -3438,6 +3446,7 @@ Create a Zstandard decompression transform.
34383446
[`zlib.createZipArchive()`]: #zlibcreateziparchiveentries-options
34393447
[`zlib.createZipArchiveSync()`]: #zlibcreateziparchivesyncentries-options
34403448
[`zlib.getMaxZipContentSize()`]: #zlibgetmaxzipcontentsize
3449+
[`zlib.zstdCompress()`]: #zlibzstdcompressbuffer-options-callback
34413450
[convenience methods]: #convenience-methods
34423451
[zlib documentation]: https://zlib.net/manual.html#Constants
34433452
[zlib.createGzip example]: #zlib

‎lib/zlib.js‎

Lines changed: 30 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -823,7 +823,7 @@ function Unzip(opts) {
823823
ObjectSetPrototypeOf(Unzip.prototype, Zlib.prototype);
824824
ObjectSetPrototypeOf(Unzip, Zlib);
825825

826-
function createConvenienceMethod(ctor, sync) {
826+
function createConvenienceMethod(ctor, sync, prepareOpts) {
827827
if (sync) {
828828
return function syncBufferWrapper(buffer, opts) {
829829
return zlibBufferSync(new ctor(opts), buffer);
@@ -834,10 +834,38 @@ function createConvenienceMethod(ctor, sync) {
834834
callback = opts;
835835
opts = {};
836836
}
837+
if (prepareOpts !== undefined) {
838+
opts = prepareOpts(buffer, opts);
839+
}
837840
return zlibBuffer(new ctor(opts), buffer, callback);
838841
};
839842
}
840843

844+
// zstdCompress() writes the input and ends the frame in separate calls, so
845+
// unlike zstdCompressSync() zstd cannot infer the input size and sizes its
846+
// tables for an unbounded stream. Pledge the size, which is known up front.
847+
function withPledgedSrcSize(buffer, opts) {
848+
if (opts?.pledgedSrcSize !== undefined) {
849+
return opts;
850+
}
851+
let pledgedSrcSize;
852+
if (typeof buffer === 'string') {
853+
// The stream encodes strings with defaultEncoding, so only a UTF-8 length
854+
// is known to match what gets written.
855+
const encoding = opts?.defaultEncoding;
856+
if (encoding != null && encoding !== 'utf8' && encoding !== 'utf-8') {
857+
return opts;
858+
}
859+
pledgedSrcSize = Buffer.byteLength(buffer);
860+
} else if (isArrayBufferView(buffer) || isAnyArrayBuffer(buffer)) {
861+
pledgedSrcSize = buffer.byteLength;
862+
} else {
863+
// Leave invalid input to the existing validation.
864+
return opts;
865+
}
866+
return { __proto__: null, ...opts, pledgedSrcSize };
867+
}
868+
841869
const kMaxBrotliParam = MathMax(
842870
...ObjectEntries(constants)
843871
.map(({ 0: key, 1: value }) => (key.startsWith('BROTLI_PARAM_') ? value : 0)),
@@ -1126,7 +1154,7 @@ module.exports = {
11261154
brotliCompressSync: createConvenienceMethod(BrotliCompress, true),
11271155
brotliDecompress: createConvenienceMethod(BrotliDecompress, false),
11281156
brotliDecompressSync: createConvenienceMethod(BrotliDecompress, true),
1129-
zstdCompress: createConvenienceMethod(ZstdCompress, false),
1157+
zstdCompress: createConvenienceMethod(ZstdCompress, false, withPledgedSrcSize),
11301158
zstdCompressSync: createConvenienceMethod(ZstdCompress, true),
11311159
zstdDecompress: createConvenienceMethod(ZstdDecompress, false),
11321160
zstdDecompressSync: createConvenienceMethod(ZstdDecompress, true),

‎test/parallel/test-zlib-zstd-pledged-src-size.js‎

Lines changed: 44 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -112,3 +112,47 @@ for (const pledgedSrcSize of [
112112
zlib.createZstdCompress({
113113
pledgedSrcSize: Number.MAX_SAFE_INTEGER,
114114
}).destroy();
115+
116+
// zstdCompress() pledges the input size by default, so its output matches
117+
// zstdCompressSync(), which lets zstd infer the size from a single call.
118+
{
119+
const text = 'héllo wörld 🚀 '.repeat(1000);
120+
const bytes = Buffer.from(text);
121+
const inputs = [
122+
'',
123+
text,
124+
bytes,
125+
new Uint16Array(bytes.buffer, bytes.byteOffset, bytes.length >> 1),
126+
new DataView(bytes.buffer, bytes.byteOffset, bytes.length),
127+
bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.length),
128+
];
129+
const opts = {
130+
params: { [zlib.constants.ZSTD_c_compressionLevel]: 9 },
131+
};
132+
133+
for (const input of inputs) {
134+
zlib.zstdCompress(input, opts, common.mustSucceed((compressed) => {
135+
assert.deepStrictEqual(compressed, zlib.zstdCompressSync(input, opts));
136+
}));
137+
}
138+
139+
for (const defaultEncoding of ['utf8', 'utf-8']) {
140+
const encodingOpts = { ...opts, defaultEncoding };
141+
zlib.zstdCompress(text, encodingOpts, common.mustSucceed((compressed) => {
142+
assert.deepStrictEqual(compressed, zlib.zstdCompressSync(text, encodingOpts));
143+
}));
144+
}
145+
146+
// The caller's options are left untouched, so they can be reused.
147+
assert.strictEqual(opts.pledgedSrcSize, undefined);
148+
149+
// An explicit pledgedSrcSize is still honored.
150+
zlib.zstdCompress(bytes, { pledgedSrcSize: 1 }, common.mustCall((err) => {
151+
assert.strictEqual(err.code, pledgedSrcSizeError.code);
152+
}));
153+
154+
// Strings written with a non-UTF-8 defaultEncoding still compress.
155+
zlib.zstdCompress('é', { defaultEncoding: 'latin1' }, common.mustSucceed((compressed) => {
156+
assert.deepStrictEqual(zlib.zstdDecompressSync(compressed), Buffer.from([0xe9]));
157+
}));
158+
}

0 commit comments

Comments
 (0)