Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
116 changes: 71 additions & 45 deletions src/jsc/modules/NodeBufferModule.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
#include <JavaScriptCore/JSTypedArrays.h>

namespace WebCore {
JSC::JSUint8Array* createBuffer(JSC::JSGlobalObject*, std::span<const uint8_t>);
JSC::JSUint8Array* createUninitializedBuffer(JSC::JSGlobalObject*, size_t);
JSC::JSUint8Array* createEmptyBuffer(JSC::JSGlobalObject*);
}

Expand Down Expand Up @@ -49,11 +49,13 @@ static bool transcodeDecodeToUtf16(std::span<const uint8_t> input, TranscodeEnco
const auto* data = reinterpret_cast<const char*>(input.data());
switch (fromEncoding) {
case TranscodeEncoding::Latin1:
units.grow(input.size());
if (!units.tryGrow(input.size())) [[unlikely]]
return false;
(void)simdutf::convert_latin1_to_utf16le(data, input.size(), units.begin());
break;
case TranscodeEncoding::Ascii: {
units.grow(input.size());
if (!units.tryGrow(input.size())) [[unlikely]]
return false;
(void)simdutf::convert_latin1_to_utf16le(data, input.size(), units.begin());
// ICU's ascii converter substitutes non-ASCII bytes with U+FFFD;
// simdutf has no substituting decode, so fix up only when needed.
Expand All @@ -71,7 +73,8 @@ static bool transcodeDecodeToUtf16(std::span<const uint8_t> input, TranscodeEnco
auto decoded = Zig::convertUTF8ToString(std::span { reinterpret_cast<const unsigned char*>(input.data()), input.size() });
if (decoded.isNull() && !input.empty()) [[unlikely]]
return false;
units.grow(decoded.length());
if (!units.tryGrow(decoded.length())) [[unlikely]]
return false;
if (decoded.is8Bit())
(void)simdutf::convert_latin1_to_utf16le(reinterpret_cast<const char*>(decoded.span8().data()), decoded.length(), units.begin());
else
Expand All @@ -83,11 +86,13 @@ static bool transcodeDecodeToUtf16(std::span<const uint8_t> input, TranscodeEnco
// the trailing odd byte of the source is dropped for narrow targets
// (Node floors the char count) but replaced for a ucs2 target.
const size_t lengthInChars = input.size() / 2;
units.grow(lengthInChars);
const bool replacesOddByte = replaceTrailingOddByte && (input.size() & 1);
if (!units.tryGrow(lengthInChars + replacesOddByte)) [[unlikely]]
return false;
memcpy(units.begin(), input.data(), lengthInChars * 2);
simdutf::to_well_formed_utf16le(units.begin(), lengthInChars, units.begin());
if (replaceTrailingOddByte && (input.size() & 1))
units.append(0xFFFD);
if (replacesOddByte)
units.last() = 0xFFFD;
break;
}
default:
Expand All @@ -98,28 +103,39 @@ static bool transcodeDecodeToUtf16(std::span<const uint8_t> input, TranscodeEnco

// Encode well-formed UTF-16 into a single-byte encoding: code points above
// maxCodePoint become '?', matching ICU's substitution behavior.
static void transcodeEncodeNarrow(const WTF::Vector<char16_t>& units, char16_t maxCodePoint, WTF::Vector<uint8_t>& out)
// Returns nullptr with an exception pending when the result cannot be allocated.
static JSC::JSUint8Array* transcodeEncodeNarrow(JSGlobalObject* globalObject, const WTF::Vector<char16_t>& units, char16_t maxCodePoint)
{
auto scope = DECLARE_THROW_SCOPE(globalObject->vm());
JSC::JSUint8Array* result = nullptr;

// Fast path: a latin1 target with in-range contents converts in bulk.
if (maxCodePoint == 0xFF) {
out.grow(units.size());
auto result = simdutf::convert_utf16le_to_latin1_with_errors(units.begin(), units.size(), reinterpret_cast<char*>(out.begin()));
if (result.error == simdutf::error_code::SUCCESS)
return;
out.shrink(0);
result = WebCore::createUninitializedBuffer(globalObject, units.size());
Comment thread
coderabbitai[bot] marked this conversation as resolved.
RETURN_IF_EXCEPTION(scope, nullptr);
auto converted = simdutf::convert_utf16le_to_latin1_with_errors(units.begin(), units.size(), reinterpret_cast<char*>(result->typedVector()));
if (converted.error == simdutf::error_code::SUCCESS)
return result;
}
// Substitution path: simdutf conversions are strict, so out-of-range
// code points ('?' in ICU) are handled per unit. `units` is well-formed,
// so a lead surrogate always has its trail: the pair is one code point.
for (size_t i = 0; i < units.size(); i++) {
const char16_t unit = units[i];
if (U16_IS_LEAD(unit)) {
out.append('?');
i++;
continue;
}
out.append(unit <= maxCodePoint ? static_cast<uint8_t>(unit) : '?');
// code points ('?' in ICU) are handled per unit.
constexpr auto writesByte = [](char16_t unit) { return !U16_IS_TRAIL(unit); };
size_t length = 0;
for (const char16_t unit : units)
length += writesByte(unit);
// A surrogate pair writes one '?', so a source with pairs needs a shorter result than the bulk attempt allocated.
if (!result || length != units.size()) {
result = WebCore::createUninitializedBuffer(globalObject, length);
RETURN_IF_EXCEPTION(scope, nullptr);
}
const std::span<uint8_t> out = result->typedSpan();
size_t written = 0;
for (const char16_t unit : units) {
if (writesByte(unit))
out[written++] = unit <= maxCodePoint ? static_cast<uint8_t>(unit) : '?';
}
ASSERT(written == out.size());
return result;
}

} // namespace
Expand Down Expand Up @@ -174,7 +190,8 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode,

int32_t errorCode = 0;
ASCIILiteral errorName;
WTF::Vector<uint8_t> result;
// An allocation that fails throws RangeError: Out of memory. Node's ICU paths have an INT32_MAX limit and report U_ILLEGAL_ARGUMENT_ERROR: https://github.com/nodejs/node/blob/v26.3.0/src/node_i18n.cc#L158-L163
JSC::JSUint8Array* result = nullptr;

if (fromEncoding == TranscodeEncoding::Unsupported || toEncoding == TranscodeEncoding::Unsupported) {
errorCode = U_ILLEGAL_ARGUMENT_ERRNO;
Expand All @@ -183,34 +200,39 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode,
&& toEncoding == TranscodeEncoding::Ucs2) {
// Node's TranscodeLatin1ToUcs2: widen each byte to a UTF-16LE unit
// (an ASCII source is treated as latin1 here, matching Node).
result.grow(length * 2);
result = WebCore::createUninitializedBuffer(globalObject, length * 2);
RETURN_IF_EXCEPTION(scope, {});
// Latin1 -> UTF-16 cannot fail; every byte is a valid code unit.
(void)simdutf::convert_latin1_to_utf16le(data, length, reinterpret_cast<char16_t*>(result.begin()));
(void)simdutf::convert_latin1_to_utf16le(data, length, reinterpret_cast<char16_t*>(result->typedVector()));
} else if (fromEncoding == TranscodeEncoding::Utf8 && toEncoding == TranscodeEncoding::Ucs2) {
// Node's TranscodeUcs2FromUtf8: invalid UTF-8 fails.
const size_t expected = simdutf::utf16_length_from_utf8(data, length);
result.grow(expected * 2);
const size_t actual = simdutf::convert_utf8_to_utf16le(data, length, reinterpret_cast<char16_t*>(result.begin()));
if (actual == 0) {
result = WebCore::createUninitializedBuffer(globalObject, expected * 2);
RETURN_IF_EXCEPTION(scope, {});
const size_t actual = simdutf::convert_utf8_to_utf16le(data, length, reinterpret_cast<char16_t*>(result->typedVector()));
// A shared source can change between the two passes. This keeps unwritten bytes out of the result; it does not bound the writes.
if (actual == 0 || actual != expected) {
Comment thread
robobun marked this conversation as resolved.
errorCode = U_INVALID_CHAR_FOUND_ERRNO;
errorName = "U_INVALID_CHAR_FOUND"_s;
} else {
result.shrink(actual * 2);
}
} else if (fromEncoding == TranscodeEncoding::Ucs2 && toEncoding == TranscodeEncoding::Utf8) {
// Node's TranscodeUtf8FromUcs2: lone surrogates fail; a trailing odd
// byte is dropped.
const size_t lengthInChars = length / 2;
WTF::Vector<char16_t> sourceBuffer(lengthInChars);
WTF::Vector<char16_t> sourceBuffer;
if (!sourceBuffer.tryGrow(lengthInChars)) [[unlikely]] {
throwOutOfMemoryError(globalObject, scope);
return {};
}
memcpy(sourceBuffer.begin(), data, lengthInChars * 2);
const size_t expected = simdutf::utf8_length_from_utf16le(sourceBuffer.begin(), lengthInChars);
result.grow(expected);
const size_t actual = simdutf::convert_utf16le_to_utf8(sourceBuffer.begin(), lengthInChars, reinterpret_cast<char*>(result.begin()));
if (actual == 0) {
result = WebCore::createUninitializedBuffer(globalObject, expected);
RETURN_IF_EXCEPTION(scope, {});
const size_t actual = simdutf::convert_utf16le_to_utf8(sourceBuffer.begin(), lengthInChars, reinterpret_cast<char*>(result->typedVector()));
// Node also fails a source of one byte, which has no units: https://github.com/nodejs/node/blob/v26.3.0/src/node_i18n.cc#L249-L252
if (actual == 0 || actual != expected) {
Comment thread
robobun marked this conversation as resolved.
errorCode = U_INVALID_CHAR_FOUND_ERRNO;
errorName = "U_INVALID_CHAR_FOUND"_s;
} else {
result.shrink(actual);
}
} else {
// Decode to well-formed UTF-16, then encode to the target.
Expand All @@ -222,21 +244,25 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode,

switch (toEncoding) {
case TranscodeEncoding::Latin1:
transcodeEncodeNarrow(units, 0xFF, result);
result = transcodeEncodeNarrow(globalObject, units, 0xFF);
RETURN_IF_EXCEPTION(scope, {});
break;
case TranscodeEncoding::Ascii:
transcodeEncodeNarrow(units, 0x7F, result);
result = transcodeEncodeNarrow(globalObject, units, 0x7F);
RETURN_IF_EXCEPTION(scope, {});
break;
case TranscodeEncoding::Ucs2:
result.grow(units.size() * 2);
memcpy(result.begin(), units.begin(), units.size() * 2);
result = WebCore::createUninitializedBuffer(globalObject, units.size() * 2);
RETURN_IF_EXCEPTION(scope, {});
memcpy(result->typedVector(), units.begin(), units.size() * 2);
break;
case TranscodeEncoding::Utf8: {
// `units` is well-formed UTF-16, so this conversion cannot fail.
const size_t expected = simdutf::utf8_length_from_utf16le(units.begin(), units.size());
result.grow(expected);
const size_t actual = simdutf::convert_utf16le_to_utf8(units.begin(), units.size(), reinterpret_cast<char*>(result.begin()));
result.shrink(actual);
result = WebCore::createUninitializedBuffer(globalObject, expected);
RETURN_IF_EXCEPTION(scope, {});
const size_t actual = simdutf::convert_utf16le_to_utf8(units.begin(), units.size(), reinterpret_cast<char*>(result->typedVector()));
RELEASE_ASSERT(actual == expected);
break;
}
default:
Expand All @@ -252,5 +278,5 @@ BUN_DEFINE_HOST_FUNCTION(jsBufferTranscode,
return {};
}

RELEASE_AND_RETURN(scope, JSValue::encode(WebCore::createBuffer(globalObject, result)));
return JSValue::encode(result);
}
151 changes: 151 additions & 0 deletions test/js/node/buffer.test.js
Original file line number Diff line number Diff line change
Expand Up @@ -5143,6 +5143,157 @@ it.skipIf(os.totalmem() < 10 * 1024 ** 3)(
},
);

// A latin1 or ascii target gets one byte for each code point: '?' when the target cannot encode it,
// and one '?' for a surrogate pair. The lengths sit on both sides of the simdutf block sizes and of
// the 1000 bytes above which a typed array is allocated with malloc.
it("transcode to latin1 and ascii writes one byte for each code point", () => {
const { transcode } = BufferModule;
const questionMark = Buffer.from("?");
for (const length of [1, 15, 16, 17, 31, 32, 33, 63, 64, 65, 1001]) {
// `length` code points in every shape.
const latin1Only = Buffer.alloc(length, "A\u00e9", "latin1").toString("latin1");
const half = length >> 1;
const shapes = {
"latin1 only": latin1Only,
"U+0100 last": latin1Only.slice(0, -1) + "\u0100",
"U+6F22 first": "\u6f22" + latin1Only.slice(1),
"U+1F600 in the middle": latin1Only.slice(0, half) + "\u{1F600}" + latin1Only.slice(half + 1),
"U+1F600 only": Buffer.alloc(length * 4, "\u{1F600}").toString(),
};
for (const [shape, text] of Object.entries(shapes)) {
// A lone surrogate in a ucs2 source decodes to U+FFFD, and the trailing odd byte is dropped.
const loneSurrogates = Buffer.concat([
Buffer.from([0x00, 0xdc]),
Buffer.from(text, "ucs2"),
Buffer.from([0x00, 0xd8, 0x41]),
]);
// With the `u` flag a surrogate pair is one match.
for (const [to, notEncodable] of [
["latin1", /[^\x00-\xff]/gu],
["ascii", /[^\x00-\x7f]/gu],
]) {
const expected = Buffer.from(text.replace(notEncodable, "?"), "latin1");
for (const from of ["ucs2", "utf8"]) {
expect(transcode(Buffer.from(text, from), from, to), `${length} x ${shape}, ${from} to ${to}`).toEqual(
expected,
);
}
expect(transcode(loneSurrogates, "ucs2", to), `${length} x ${shape}, lone surrogates to ${to}`).toEqual(
Buffer.concat([questionMark, expected, questionMark]),
);
}
}
}
});

// transcode() sized its result, and the UTF-16 copy of the source it converts through, with
// WTF::Vector::grow(). That aborts the process when the allocation fails or passes 2**31 bytes, also
// inside try/catch. Each case runs in a child and prints its line as soon as it finishes, so an abort
// shows which case it was.
describe("transcode allocation limits", () => {
async function runCases(defineCases, env = {}) {
const script = `
const { transcode } = require("node:buffer");
${defineCases}
for (const [name, run] of Object.entries(cases)) {
try {
const result = run();
console.log(name + ": " + JSON.stringify({ length: result.length, head: [...result.subarray(0, 4)], tail: [...result.subarray(-4)] }));
} catch (e) {
console.log(name + ": " + e.name + ": " + e.message);
}
}
`;
await using proc = Bun.spawn({
cmd: [bunExe(), "-e", script],
env: { ...bunEnv, BUN_GARBAGE_COLLECTOR_LEVEL: "0", ...env },
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);
return { stdout: stdout.trim().split("\n"), stderr, exitCode };
}

// `BUN_JSC_maxSingleAllocationSize` exists in debug WTF only. It makes every fallible WTF allocation
// above the cap return null and every infallible one assert. A 3 MiB source decodes to a 6 MiB
// UTF-16 copy, and a 10 MiB ucs2 source to a 10 MiB one.
it.skipIf(!isDebug)("throws when the UTF-16 copy of the source cannot be allocated", async () => {
const result = await runCases(
`
const bytes = Buffer.alloc(${3 * 1024 ** 2}, 97);
const units = Buffer.alloc(${10 * 1024 ** 2}, 97);
const cases = {
"latin1 to utf8": () => transcode(bytes, "latin1", "utf8"),
"ascii to utf8": () => transcode(bytes, "ascii", "utf8"),
"utf8 to latin1": () => transcode(bytes, "utf8", "latin1"),
"ucs2 to latin1": () => transcode(units, "ucs2", "latin1"),
"ucs2 to utf8": () => transcode(units, "ucs2", "utf8"),
// Under the cap, so the copy is made.
"1 MiB of utf8 to latin1": () => transcode(bytes.subarray(0, ${1024 ** 2}), "utf8", "latin1"),
};
`,
{ BUN_JSC_maxSingleAllocationSize: String(4 * 1024 ** 2) },
);
expect(result).toEqual({
stdout: [
"latin1 to utf8: RangeError: Out of memory",
"ascii to utf8: RangeError: Out of memory",
"utf8 to latin1: RangeError: Out of memory",
"ucs2 to latin1: RangeError: Out of memory",
"ucs2 to utf8: RangeError: Out of memory",
'1 MiB of utf8 to latin1: {"length":1048576,"head":[97,97,97,97],"tail":[97,97,97,97]}',
],
stderr: "",
exitCode: 0,
});
});

// The child never writes to the source, so it stays untouched address space and the test is cheap.
// A small host can still refuse to reserve the 2 GiB.
it.skipIf(os.totalmem() < 4 * 1024 ** 3)("throws past the size limit of a Buffer or a Vector", async () => {
const result = await runCases(`
const memory = new ArrayBuffer(2 ** 31 + 1);
const cases = {
// The result needs 2**32 + 2 bytes and a Buffer holds 2**32.
"latin1 to ucs2": () => transcode(new Uint8Array(memory), "latin1", "ucs2"),
// These convert through a UTF-16 copy of 2**30 units, which is 2**31 bytes.
"latin1 to utf8": () => transcode(new Uint8Array(memory, 0, 2 ** 30), "latin1", "utf8"),
"ascii to utf8": () => transcode(new Uint8Array(memory, 0, 2 ** 30), "ascii", "utf8"),
"ucs2 to latin1": () => transcode(new Uint8Array(memory, 0, 2 ** 31), "ucs2", "latin1"),
"ucs2 to ucs2": () => transcode(new Uint8Array(memory, 0, 2 ** 31 - 1), "ucs2", "ucs2"),
"ucs2 to utf8": () => transcode(new Uint8Array(memory, 0, 2 ** 31), "ucs2", "utf8"),
};
`);
expect(result).toEqual({
stdout: [
"latin1 to ucs2: RangeError: Out of memory",
"latin1 to utf8: RangeError: Out of memory",
"ascii to utf8: RangeError: Out of memory",
"ucs2 to latin1: RangeError: Out of memory",
"ucs2 to ucs2: RangeError: Out of memory",
"ucs2 to utf8: RangeError: Out of memory",
],
stderr: "",
exitCode: 0,
});
});

// Node v26.3.0 returns the same 2 GiB Buffer. The child writes all of it.
it.skipIf(os.totalmem() < 10 * 1024 ** 3)("returns a result of 2 GiB", async () => {
const result = await runCases(`
const source = new Uint8Array(2 ** 30);
source[0] = 0xe9;
source[source.length - 1] = 0x41;
const cases = { "latin1 to ucs2": () => transcode(source, "latin1", "ucs2") };
`);
expect(result).toEqual({
stdout: ['latin1 to ucs2: {"length":2147483648,"head":[233,0,0,0],"tail":[0,0,65,0]}'],
stderr: "",
exitCode: 0,
});
});
});

// The fixed-width read* / write* accessors are C++ host functions that JSC's DFG/FTL compile into
// bounds-checked loads / stores (JSBuffer.cpp + JavaScriptCore's BufferAccessorRegistry). They must
// keep agreeing with a DataView reference after tier-up, and everything the JIT does not speculate
Expand Down
Loading