diff --git a/src/jsc/modules/NodeBufferModule.cpp b/src/jsc/modules/NodeBufferModule.cpp index 246da9513dbf..50ec39550c81 100644 --- a/src/jsc/modules/NodeBufferModule.cpp +++ b/src/jsc/modules/NodeBufferModule.cpp @@ -9,7 +9,7 @@ #include namespace WebCore { -JSC::JSUint8Array* createBuffer(JSC::JSGlobalObject*, std::span); +JSC::JSUint8Array* createUninitializedBuffer(JSC::JSGlobalObject*, size_t); JSC::JSUint8Array* createEmptyBuffer(JSC::JSGlobalObject*); } @@ -49,11 +49,13 @@ static bool transcodeDecodeToUtf16(std::span input, TranscodeEnco const auto* data = reinterpret_cast(input.data()); switch (fromEncoding) { case TranscodeEncoding::Latin1: - units.grow(input.size()); + if (!units.tryGrow(input.size())) [[unlikely]] + return false; (void)simdutf::convert_latin1_to_utf16le(data, input.size(), units.begin()); break; case TranscodeEncoding::Ascii: { - units.grow(input.size()); + if (!units.tryGrow(input.size())) [[unlikely]] + return false; (void)simdutf::convert_latin1_to_utf16le(data, input.size(), units.begin()); // ICU's ascii converter substitutes non-ASCII bytes with U+FFFD; // simdutf has no substituting decode, so fix up only when needed. @@ -71,7 +73,8 @@ static bool transcodeDecodeToUtf16(std::span input, TranscodeEnco auto decoded = Zig::convertUTF8ToString(std::span { reinterpret_cast(input.data()), input.size() }); if (decoded.isNull() && !input.empty()) [[unlikely]] return false; - units.grow(decoded.length()); + if (!units.tryGrow(decoded.length())) [[unlikely]] + return false; if (decoded.is8Bit()) (void)simdutf::convert_latin1_to_utf16le(reinterpret_cast(decoded.span8().data()), decoded.length(), units.begin()); else @@ -83,11 +86,13 @@ static bool transcodeDecodeToUtf16(std::span input, TranscodeEnco // the trailing odd byte of the source is dropped for narrow targets // (Node floors the char count) but replaced for a ucs2 target. const size_t lengthInChars = input.size() / 2; - units.grow(lengthInChars); + const bool replacesOddByte = replaceTrailingOddByte && (input.size() & 1); + if (!units.tryGrow(lengthInChars + replacesOddByte)) [[unlikely]] + return false; memcpy(units.begin(), input.data(), lengthInChars * 2); simdutf::to_well_formed_utf16le(units.begin(), lengthInChars, units.begin()); - if (replaceTrailingOddByte && (input.size() & 1)) - units.append(0xFFFD); + if (replacesOddByte) + units.last() = 0xFFFD; break; } default: @@ -98,28 +103,39 @@ static bool transcodeDecodeToUtf16(std::span input, TranscodeEnco // Encode well-formed UTF-16 into a single-byte encoding: code points above // maxCodePoint become '?', matching ICU's substitution behavior. -static void transcodeEncodeNarrow(const WTF::Vector& units, char16_t maxCodePoint, WTF::Vector& out) +// Returns nullptr with an exception pending when the result cannot be allocated. +static JSC::JSUint8Array* transcodeEncodeNarrow(JSGlobalObject* globalObject, const WTF::Vector& units, char16_t maxCodePoint) { + auto scope = DECLARE_THROW_SCOPE(globalObject->vm()); + JSC::JSUint8Array* result = nullptr; + // Fast path: a latin1 target with in-range contents converts in bulk. if (maxCodePoint == 0xFF) { - out.grow(units.size()); - auto result = simdutf::convert_utf16le_to_latin1_with_errors(units.begin(), units.size(), reinterpret_cast(out.begin())); - if (result.error == simdutf::error_code::SUCCESS) - return; - out.shrink(0); + result = WebCore::createUninitializedBuffer(globalObject, units.size()); + RETURN_IF_EXCEPTION(scope, nullptr); + auto converted = simdutf::convert_utf16le_to_latin1_with_errors(units.begin(), units.size(), reinterpret_cast(result->typedVector())); + if (converted.error == simdutf::error_code::SUCCESS) + return result; } // Substitution path: simdutf conversions are strict, so out-of-range - // code points ('?' in ICU) are handled per unit. `units` is well-formed, - // so a lead surrogate always has its trail: the pair is one code point. - for (size_t i = 0; i < units.size(); i++) { - const char16_t unit = units[i]; - if (U16_IS_LEAD(unit)) { - out.append('?'); - i++; - continue; - } - out.append(unit <= maxCodePoint ? static_cast(unit) : '?'); + // code points ('?' in ICU) are handled per unit. + constexpr auto writesByte = [](char16_t unit) { return !U16_IS_TRAIL(unit); }; + size_t length = 0; + for (const char16_t unit : units) + length += writesByte(unit); + // A surrogate pair writes one '?', so a source with pairs needs a shorter result than the bulk attempt allocated. + if (!result || length != units.size()) { + result = WebCore::createUninitializedBuffer(globalObject, length); + RETURN_IF_EXCEPTION(scope, nullptr); + } + const std::span out = result->typedSpan(); + size_t written = 0; + for (const char16_t unit : units) { + if (writesByte(unit)) + out[written++] = unit <= maxCodePoint ? static_cast(unit) : '?'; } + ASSERT(written == out.size()); + return result; } } // namespace @@ -174,7 +190,8 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode, int32_t errorCode = 0; ASCIILiteral errorName; - WTF::Vector result; + // An allocation that fails throws RangeError: Out of memory. Node's ICU paths have an INT32_MAX limit and report U_ILLEGAL_ARGUMENT_ERROR: https://github.com/nodejs/node/blob/v26.3.0/src/node_i18n.cc#L158-L163 + JSC::JSUint8Array* result = nullptr; if (fromEncoding == TranscodeEncoding::Unsupported || toEncoding == TranscodeEncoding::Unsupported) { errorCode = U_ILLEGAL_ARGUMENT_ERRNO; @@ -183,34 +200,39 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode, && toEncoding == TranscodeEncoding::Ucs2) { // Node's TranscodeLatin1ToUcs2: widen each byte to a UTF-16LE unit // (an ASCII source is treated as latin1 here, matching Node). - result.grow(length * 2); + result = WebCore::createUninitializedBuffer(globalObject, length * 2); + RETURN_IF_EXCEPTION(scope, {}); // Latin1 -> UTF-16 cannot fail; every byte is a valid code unit. - (void)simdutf::convert_latin1_to_utf16le(data, length, reinterpret_cast(result.begin())); + (void)simdutf::convert_latin1_to_utf16le(data, length, reinterpret_cast(result->typedVector())); } else if (fromEncoding == TranscodeEncoding::Utf8 && toEncoding == TranscodeEncoding::Ucs2) { // Node's TranscodeUcs2FromUtf8: invalid UTF-8 fails. const size_t expected = simdutf::utf16_length_from_utf8(data, length); - result.grow(expected * 2); - const size_t actual = simdutf::convert_utf8_to_utf16le(data, length, reinterpret_cast(result.begin())); - if (actual == 0) { + result = WebCore::createUninitializedBuffer(globalObject, expected * 2); + RETURN_IF_EXCEPTION(scope, {}); + const size_t actual = simdutf::convert_utf8_to_utf16le(data, length, reinterpret_cast(result->typedVector())); + // A shared source can change between the two passes. This keeps unwritten bytes out of the result; it does not bound the writes. + if (actual == 0 || actual != expected) { errorCode = U_INVALID_CHAR_FOUND_ERRNO; errorName = "U_INVALID_CHAR_FOUND"_s; - } else { - result.shrink(actual * 2); } } else if (fromEncoding == TranscodeEncoding::Ucs2 && toEncoding == TranscodeEncoding::Utf8) { // Node's TranscodeUtf8FromUcs2: lone surrogates fail; a trailing odd // byte is dropped. const size_t lengthInChars = length / 2; - WTF::Vector sourceBuffer(lengthInChars); + WTF::Vector sourceBuffer; + if (!sourceBuffer.tryGrow(lengthInChars)) [[unlikely]] { + throwOutOfMemoryError(globalObject, scope); + return {}; + } memcpy(sourceBuffer.begin(), data, lengthInChars * 2); const size_t expected = simdutf::utf8_length_from_utf16le(sourceBuffer.begin(), lengthInChars); - result.grow(expected); - const size_t actual = simdutf::convert_utf16le_to_utf8(sourceBuffer.begin(), lengthInChars, reinterpret_cast(result.begin())); - if (actual == 0) { + result = WebCore::createUninitializedBuffer(globalObject, expected); + RETURN_IF_EXCEPTION(scope, {}); + const size_t actual = simdutf::convert_utf16le_to_utf8(sourceBuffer.begin(), lengthInChars, reinterpret_cast(result->typedVector())); + // Node also fails a source of one byte, which has no units: https://github.com/nodejs/node/blob/v26.3.0/src/node_i18n.cc#L249-L252 + if (actual == 0 || actual != expected) { errorCode = U_INVALID_CHAR_FOUND_ERRNO; errorName = "U_INVALID_CHAR_FOUND"_s; - } else { - result.shrink(actual); } } else { // Decode to well-formed UTF-16, then encode to the target. @@ -222,21 +244,25 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode, switch (toEncoding) { case TranscodeEncoding::Latin1: - transcodeEncodeNarrow(units, 0xFF, result); + result = transcodeEncodeNarrow(globalObject, units, 0xFF); + RETURN_IF_EXCEPTION(scope, {}); break; case TranscodeEncoding::Ascii: - transcodeEncodeNarrow(units, 0x7F, result); + result = transcodeEncodeNarrow(globalObject, units, 0x7F); + RETURN_IF_EXCEPTION(scope, {}); break; case TranscodeEncoding::Ucs2: - result.grow(units.size() * 2); - memcpy(result.begin(), units.begin(), units.size() * 2); + result = WebCore::createUninitializedBuffer(globalObject, units.size() * 2); + RETURN_IF_EXCEPTION(scope, {}); + memcpy(result->typedVector(), units.begin(), units.size() * 2); break; case TranscodeEncoding::Utf8: { // `units` is well-formed UTF-16, so this conversion cannot fail. const size_t expected = simdutf::utf8_length_from_utf16le(units.begin(), units.size()); - result.grow(expected); - const size_t actual = simdutf::convert_utf16le_to_utf8(units.begin(), units.size(), reinterpret_cast(result.begin())); - result.shrink(actual); + result = WebCore::createUninitializedBuffer(globalObject, expected); + RETURN_IF_EXCEPTION(scope, {}); + const size_t actual = simdutf::convert_utf16le_to_utf8(units.begin(), units.size(), reinterpret_cast(result->typedVector())); + RELEASE_ASSERT(actual == expected); break; } default: @@ -252,5 +278,5 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode, return {}; } - RELEASE_AND_RETURN(scope, JSValue::encode(WebCore::createBuffer(globalObject, result))); + return JSValue::encode(result); } diff --git a/test/js/node/buffer.test.js b/test/js/node/buffer.test.js index 5b4f065be33f..01ad7373d979 100644 --- a/test/js/node/buffer.test.js +++ b/test/js/node/buffer.test.js @@ -5143,6 +5143,157 @@ it.skipIf(os.totalmem() < 10 * 1024 ** 3)( }, ); +// A latin1 or ascii target gets one byte for each code point: '?' when the target cannot encode it, +// and one '?' for a surrogate pair. The lengths sit on both sides of the simdutf block sizes and of +// the 1000 bytes above which a typed array is allocated with malloc. +it("transcode to latin1 and ascii writes one byte for each code point", () => { + const { transcode } = BufferModule; + const questionMark = Buffer.from("?"); + for (const length of [1, 15, 16, 17, 31, 32, 33, 63, 64, 65, 1001]) { + // `length` code points in every shape. + const latin1Only = Buffer.alloc(length, "A\u00e9", "latin1").toString("latin1"); + const half = length >> 1; + const shapes = { + "latin1 only": latin1Only, + "U+0100 last": latin1Only.slice(0, -1) + "\u0100", + "U+6F22 first": "\u6f22" + latin1Only.slice(1), + "U+1F600 in the middle": latin1Only.slice(0, half) + "\u{1F600}" + latin1Only.slice(half + 1), + "U+1F600 only": Buffer.alloc(length * 4, "\u{1F600}").toString(), + }; + for (const [shape, text] of Object.entries(shapes)) { + // A lone surrogate in a ucs2 source decodes to U+FFFD, and the trailing odd byte is dropped. + const loneSurrogates = Buffer.concat([ + Buffer.from([0x00, 0xdc]), + Buffer.from(text, "ucs2"), + Buffer.from([0x00, 0xd8, 0x41]), + ]); + // With the `u` flag a surrogate pair is one match. + for (const [to, notEncodable] of [ + ["latin1", /[^\x00-\xff]/gu], + ["ascii", /[^\x00-\x7f]/gu], + ]) { + const expected = Buffer.from(text.replace(notEncodable, "?"), "latin1"); + for (const from of ["ucs2", "utf8"]) { + expect(transcode(Buffer.from(text, from), from, to), `${length} x ${shape}, ${from} to ${to}`).toEqual( + expected, + ); + } + expect(transcode(loneSurrogates, "ucs2", to), `${length} x ${shape}, lone surrogates to ${to}`).toEqual( + Buffer.concat([questionMark, expected, questionMark]), + ); + } + } + } +}); + +// transcode() sized its result, and the UTF-16 copy of the source it converts through, with +// WTF::Vector::grow(). That aborts the process when the allocation fails or passes 2**31 bytes, also +// inside try/catch. Each case runs in a child and prints its line as soon as it finishes, so an abort +// shows which case it was. +describe("transcode allocation limits", () => { + async function runCases(defineCases, env = {}) { + const script = ` + const { transcode } = require("node:buffer"); + ${defineCases} + for (const [name, run] of Object.entries(cases)) { + try { + const result = run(); + console.log(name + ": " + JSON.stringify({ length: result.length, head: [...result.subarray(0, 4)], tail: [...result.subarray(-4)] })); + } catch (e) { + console.log(name + ": " + e.name + ": " + e.message); + } + } + `; + await using proc = Bun.spawn({ + cmd: [bunExe(), "-e", script], + env: { ...bunEnv, BUN_GARBAGE_COLLECTOR_LEVEL: "0", ...env }, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]); + return { stdout: stdout.trim().split("\n"), stderr, exitCode }; + } + + // `BUN_JSC_maxSingleAllocationSize` exists in debug WTF only. It makes every fallible WTF allocation + // above the cap return null and every infallible one assert. A 3 MiB source decodes to a 6 MiB + // UTF-16 copy, and a 10 MiB ucs2 source to a 10 MiB one. + it.skipIf(!isDebug)("throws when the UTF-16 copy of the source cannot be allocated", async () => { + const result = await runCases( + ` + const bytes = Buffer.alloc(${3 * 1024 ** 2}, 97); + const units = Buffer.alloc(${10 * 1024 ** 2}, 97); + const cases = { + "latin1 to utf8": () => transcode(bytes, "latin1", "utf8"), + "ascii to utf8": () => transcode(bytes, "ascii", "utf8"), + "utf8 to latin1": () => transcode(bytes, "utf8", "latin1"), + "ucs2 to latin1": () => transcode(units, "ucs2", "latin1"), + "ucs2 to utf8": () => transcode(units, "ucs2", "utf8"), + // Under the cap, so the copy is made. + "1 MiB of utf8 to latin1": () => transcode(bytes.subarray(0, ${1024 ** 2}), "utf8", "latin1"), + }; + `, + { BUN_JSC_maxSingleAllocationSize: String(4 * 1024 ** 2) }, + ); + expect(result).toEqual({ + stdout: [ + "latin1 to utf8: RangeError: Out of memory", + "ascii to utf8: RangeError: Out of memory", + "utf8 to latin1: RangeError: Out of memory", + "ucs2 to latin1: RangeError: Out of memory", + "ucs2 to utf8: RangeError: Out of memory", + '1 MiB of utf8 to latin1: {"length":1048576,"head":[97,97,97,97],"tail":[97,97,97,97]}', + ], + stderr: "", + exitCode: 0, + }); + }); + + // The child never writes to the source, so it stays untouched address space and the test is cheap. + // A small host can still refuse to reserve the 2 GiB. + it.skipIf(os.totalmem() < 4 * 1024 ** 3)("throws past the size limit of a Buffer or a Vector", async () => { + const result = await runCases(` + const memory = new ArrayBuffer(2 ** 31 + 1); + const cases = { + // The result needs 2**32 + 2 bytes and a Buffer holds 2**32. + "latin1 to ucs2": () => transcode(new Uint8Array(memory), "latin1", "ucs2"), + // These convert through a UTF-16 copy of 2**30 units, which is 2**31 bytes. + "latin1 to utf8": () => transcode(new Uint8Array(memory, 0, 2 ** 30), "latin1", "utf8"), + "ascii to utf8": () => transcode(new Uint8Array(memory, 0, 2 ** 30), "ascii", "utf8"), + "ucs2 to latin1": () => transcode(new Uint8Array(memory, 0, 2 ** 31), "ucs2", "latin1"), + "ucs2 to ucs2": () => transcode(new Uint8Array(memory, 0, 2 ** 31 - 1), "ucs2", "ucs2"), + "ucs2 to utf8": () => transcode(new Uint8Array(memory, 0, 2 ** 31), "ucs2", "utf8"), + }; + `); + expect(result).toEqual({ + stdout: [ + "latin1 to ucs2: RangeError: Out of memory", + "latin1 to utf8: RangeError: Out of memory", + "ascii to utf8: RangeError: Out of memory", + "ucs2 to latin1: RangeError: Out of memory", + "ucs2 to ucs2: RangeError: Out of memory", + "ucs2 to utf8: RangeError: Out of memory", + ], + stderr: "", + exitCode: 0, + }); + }); + + // Node v26.3.0 returns the same 2 GiB Buffer. The child writes all of it. + it.skipIf(os.totalmem() < 10 * 1024 ** 3)("returns a result of 2 GiB", async () => { + const result = await runCases(` + const source = new Uint8Array(2 ** 30); + source[0] = 0xe9; + source[source.length - 1] = 0x41; + const cases = { "latin1 to ucs2": () => transcode(source, "latin1", "ucs2") }; + `); + expect(result).toEqual({ + stdout: ['latin1 to ucs2: {"length":2147483648,"head":[233,0,0,0],"tail":[0,0,65,0]}'], + stderr: "", + exitCode: 0, + }); + }); +}); + // The fixed-width read* / write* accessors are C++ host functions that JSC's DFG/FTL compile into // bounds-checked loads / stores (JSBuffer.cpp + JavaScriptCore's BufferAccessorRegistry). They must // keep agreeing with a DataView reference after tier-up, and everything the JIT does not speculate