diff --git a/benchmark/zlib/zstd-compress.js b/benchmark/zlib/zstd-compress.js new file mode 100644 index 00000000000..b898397c222 --- /dev/null +++ b/benchmark/zlib/zstd-compress.js @@ -0,0 +1,41 @@ +'use strict'; +const common = require('../common.js'); +const fs = require('fs'); +const zlib = require('zlib'); + +const bench = common.createBenchmark(main, { + method: ['zstdCompress', 'zstdCompressSync'], + level: [3, 12], + inputLen: [16 * 1024, 256 * 1024], + n: [100], +}); + +function main({ n, method, level, inputLen }) { + const input = Buffer.alloc(inputLen, fs.readFileSync(__filename)); + const opts = { params: { [zlib.constants.ZSTD_c_compressionLevel]: level } }; + + switch (method) { + // Performs `n` single zstdCompress operations + case 'zstdCompress': { + let i = 0; + bench.start(); + (function next(err) { + if (err) throw err; + if (i++ === n) + return bench.end(n); + zlib.zstdCompress(input, opts, next); + })(); + break; + } + // Performs `n` single zstdCompressSync operations + case 'zstdCompressSync': { + bench.start(); + for (let i = 0; i < n; ++i) + zlib.zstdCompressSync(input, opts); + bench.end(n); + break; + } + default: + throw new Error('Unsupported zstd method'); + } +} diff --git a/doc/api/zlib.md b/doc/api/zlib.md index cef9e4b7937..50401f05776 100644 --- a/doc/api/zlib.md +++ b/doc/api/zlib.md @@ -788,6 +788,9 @@ It's possible to specify the expected total size of the uncompressed input via doesn't match at the end of the input, compression will fail with the code `ZSTD_error_srcSize_wrong`. +[`zlib.zstdCompress()`][] defaults `opts.pledgedSrcSize` to the byte length of +its input. + #### Decompressor options These advanced options are available for controlling decompression: @@ -3072,6 +3075,11 @@ Decompress a chunk of data with [`Unzip`][]. added: - v23.8.0 - v22.15.0 +changes: + - version: REPLACEME + pr-url: https://github.com/nodejs/node/pull/66358 + description: The `pledgedSrcSize` option defaults to the byte length of + `buffer`. --> * `buffer` {Buffer|TypedArray|DataView|ArrayBuffer|string} @@ -3446,6 +3454,7 @@ Create a Zstandard decompression transform. [`zlib.createZipArchive()`]: #zlibcreateziparchiveentries-options [`zlib.createZipArchiveSync()`]: #zlibcreateziparchivesyncentries-options [`zlib.getMaxZipContentSize()`]: #zlibgetmaxzipcontentsize +[`zlib.zstdCompress()`]: #zlibzstdcompressbuffer-options-callback [convenience methods]: #convenience-methods [zlib documentation]: https://zlib.net/manual.html#Constants [zlib.createGzip example]: #zlib diff --git a/lib/zlib.js b/lib/zlib.js index 4a2f488835e..f6c5b177b58 100644 --- a/lib/zlib.js +++ b/lib/zlib.js @@ -789,7 +789,7 @@ class Unzip extends Zlib { } } -function createConvenienceMethod(ctor, sync) { +function createConvenienceMethod(ctor, sync, prepareOpts) { if (sync) { return function syncBufferWrapper(buffer, opts) { return zlibBufferSync(new ctor(opts), buffer); @@ -800,10 +800,38 @@ function createConvenienceMethod(ctor, sync) { callback = opts; opts = {}; } + if (prepareOpts !== undefined) { + opts = prepareOpts(buffer, opts); + } return zlibBuffer(new ctor(opts), buffer, callback); }; } +// zstdCompress() writes the input and ends the frame in separate calls, so +// unlike zstdCompressSync() zstd cannot infer the input size and sizes its +// tables for an unbounded stream. Pledge the size, which is known up front. +function withPledgedSrcSize(buffer, opts) { + if (opts?.pledgedSrcSize !== undefined) { + return opts; + } + let pledgedSrcSize; + if (typeof buffer === 'string') { + // The stream encodes strings with defaultEncoding, so only a UTF-8 length + // is known to match what gets written. + const encoding = opts?.defaultEncoding; + if (encoding != null && encoding !== 'utf8' && encoding !== 'utf-8') { + return opts; + } + pledgedSrcSize = Buffer.byteLength(buffer); + } else if (isArrayBufferView(buffer) || isAnyArrayBuffer(buffer)) { + pledgedSrcSize = buffer.byteLength; + } else { + // Leave invalid input to the existing validation. + return opts; + } + return { __proto__: null, ...opts, pledgedSrcSize }; +} + const kMaxBrotliParam = MathMax( ...ObjectEntries(constants) .map(({ 0: key, 1: value }) => (key.startsWith('BROTLI_PARAM_') ? value : 0)), @@ -1088,7 +1116,7 @@ module.exports = { brotliCompressSync: createConvenienceMethod(BrotliCompress, true), brotliDecompress: createConvenienceMethod(BrotliDecompress, false), brotliDecompressSync: createConvenienceMethod(BrotliDecompress, true), - zstdCompress: createConvenienceMethod(ZstdCompress, false), + zstdCompress: createConvenienceMethod(ZstdCompress, false, withPledgedSrcSize), zstdCompressSync: createConvenienceMethod(ZstdCompress, true), zstdDecompress: createConvenienceMethod(ZstdDecompress, false), zstdDecompressSync: createConvenienceMethod(ZstdDecompress, true), diff --git a/test/parallel/test-zlib-zstd-pledged-src-size.js b/test/parallel/test-zlib-zstd-pledged-src-size.js index a0b1babd3f7..d30a4aadefd 100644 --- a/test/parallel/test-zlib-zstd-pledged-src-size.js +++ b/test/parallel/test-zlib-zstd-pledged-src-size.js @@ -112,3 +112,47 @@ for (const pledgedSrcSize of [ zlib.createZstdCompress({ pledgedSrcSize: Number.MAX_SAFE_INTEGER, }).destroy(); + +// zstdCompress() pledges the input size by default, so its output matches +// zstdCompressSync(), which lets zstd infer the size from a single call. +{ + const text = 'héllo wörld 🚀 '.repeat(1000); + const bytes = Buffer.from(text); + const inputs = [ + '', + text, + bytes, + new Uint16Array(bytes.buffer, bytes.byteOffset, bytes.length >> 1), + new DataView(bytes.buffer, bytes.byteOffset, bytes.length), + bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.length), + ]; + const opts = { + params: { [zlib.constants.ZSTD_c_compressionLevel]: 9 }, + }; + + for (const input of inputs) { + zlib.zstdCompress(input, opts, common.mustSucceed((compressed) => { + assert.deepStrictEqual(compressed, zlib.zstdCompressSync(input, opts)); + })); + } + + for (const defaultEncoding of ['utf8', 'utf-8']) { + const encodingOpts = { ...opts, defaultEncoding }; + zlib.zstdCompress(text, encodingOpts, common.mustSucceed((compressed) => { + assert.deepStrictEqual(compressed, zlib.zstdCompressSync(text, encodingOpts)); + })); + } + + // The caller's options are left untouched, so they can be reused. + assert.strictEqual(opts.pledgedSrcSize, undefined); + + // An explicit pledgedSrcSize is still honored. + zlib.zstdCompress(bytes, { pledgedSrcSize: 1 }, common.mustCall((err) => { + assert.strictEqual(err.code, pledgedSrcSizeError.code); + })); + + // Strings written with a non-UTF-8 defaultEncoding still compress. + zlib.zstdCompress('é', { defaultEncoding: 'latin1' }, common.mustSucceed((compressed) => { + assert.deepStrictEqual(zlib.zstdDecompressSync(compressed), Buffer.from([0xe9])); + })); +}