From 05ca64d3e94a26f36f862f58919be91f1cc9a767 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:30:50 +0000 Subject: [PATCH 1/4] streams: join the string chunks of a text consumer with one allocation convertChunksToText joined a stream of string chunks in a WTF::StringBuilder with the default overflow policy, which aborts the process when it cannot grow. It reserved the sum of the lengths as an 8-bit buffer, so the first 16-bit chunk made it allocate a 16-bit buffer of twice that size. The length and the width of the text are known before the join. The consumer now makes one allocation of that length and width, and throws RangeError: Out of memory when the allocation fails. --- .../webcore/streams/BunStreamConsumers.cpp | 38 +++++--- .../web/streams/streams-string-limit.test.ts | 87 ++++++++++++++++++- test/js/web/streams/streams.test.js | 16 ++++ 3 files changed, 128 insertions(+), 13 deletions(-) diff --git a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp index f510e1a76142..58ea99874c4e 100644 --- a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp +++ b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp @@ -540,6 +540,24 @@ static JSValue convertChunksToBytes(JSGlobalObject* globalObject, JSValue chunks static JSValue textAccumulatorWrite(JSC::VM& vm, JSGlobalObject*, JSC::JSObject* owner, BunTextAccumulator&, JSValue chunk); static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject*, JSC::JSObject* owner, BunTextAccumulator&); +// The string chunks as one string of `length` code units, from one allocation. Null when that allocation fails. +template +static WTF::String tryJoinStringChunks(JSGlobalObject* globalObject, const MarkedArgumentBuffer& chunks, unsigned length) +{ + auto scope = DECLARE_THROW_SCOPE(getVM(globalObject)); + std::span characters; + WTF::String joined = WTF::String::tryCreateUninitialized(length, characters); + if (joined.isNull()) [[unlikely]] + return joined; + for (unsigned i = 0; i < chunks.size(); i++) { + auto chunk = asString(chunks.at(i))->view(globalObject); + RETURN_IF_EXCEPTION(scope, {}); + chunk->getCharacters(characters); + characters = characters.subspan(chunk->length()); + } + return joined; +} + // The chunk-array -> text conversion: pure-string arrays join once (no UTF-8 round trip); // mixed/binary chunk arrays run through the shared text accumulator. static JSValue convertChunksToText(JSGlobalObject* globalObject, JSValue chunksValue) @@ -585,6 +603,7 @@ static JSValue convertChunksToText(JSGlobalObject* globalObject, JSValue chunksV // MarkedArgumentBuffer for the conversion below. MarkedArgumentBuffer values; bool allStrings = true; + bool all8Bit = true; WTF::CheckedUint32 codeUnits = 0; for (unsigned i = 0; i < length; i++) { JSValue chunk = chunks->getIndex(globalObject, i); @@ -592,8 +611,10 @@ static JSValue convertChunksToText(JSGlobalObject* globalObject, JSValue chunksV values.append(chunk); if (!chunk.isString()) allStrings = false; - else if (allStrings) + else if (allStrings) { codeUnits += asString(chunk)->length(); + all8Bit = all8Bit && asString(chunk)->is8Bit(); + } } if (values.hasOverflowed()) [[unlikely]] { throwOutOfMemoryError(globalObject, scope); @@ -604,18 +625,15 @@ static JSValue convertChunksToText(JSGlobalObject* globalObject, JSValue chunksV throwOutOfMemoryError(globalObject, scope); return {}; } - WTF::StringBuilder rope; - rope.reserveCapacity(codeUnits.value()); - for (unsigned i = 0; i < length; i++) { - WTF::String string = asString(values.at(i))->value(globalObject); - RETURN_IF_EXCEPTION(scope, {}); - rope.append(string); - } - if (rope.hasOverflowed()) [[unlikely]] { + WTF::String joined = all8Bit + ? tryJoinStringChunks(globalObject, values, codeUnits.value()) + : tryJoinStringChunks(globalObject, values, codeUnits.value()); + RETURN_IF_EXCEPTION(scope, {}); + if (joined.isNull()) [[unlikely]] { throwOutOfMemoryError(globalObject, scope); return {}; } - RELEASE_AND_RETURN(scope, jsString(vm, stripTextResultBOM(rope.toString()))); + RELEASE_AND_RETURN(scope, jsString(vm, stripTextResultBOM(joined))); } // Mixed string/binary chunks: drive the shared accumulator so adjacent-string rope diff --git a/test/js/web/streams/streams-string-limit.test.ts b/test/js/web/streams/streams-string-limit.test.ts index 956b81189b62..643b9661fd76 100644 --- a/test/js/web/streams/streams-string-limit.test.ts +++ b/test/js/web/streams/streams-string-limit.test.ts @@ -1,5 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { bunEnv, bunExe } from "harness"; +import { bunEnv, bunExe, isASAN } from "harness"; import { totalmem } from "node:os"; // Consuming a stream as text must reject with a catchable error when the accumulated @@ -21,10 +21,13 @@ function consumeToText(streamSource: string): string { `; } -async function run(script: string): Promise<{ stdout: string; stderr: string; exitCode: number }> { +async function run( + script: string, + env: Record = bunEnv, +): Promise<{ stdout: string; stderr: string; exitCode: number }> { await using proc = Bun.spawn({ cmd: [bunExe(), "-e", script], - env: bunEnv, + env, stdout: "pipe", stderr: "pipe", }); @@ -184,6 +187,84 @@ test.skipIf(!enoughMemory)("arrayBuffer() and bytes() reject mixed chunks summin }); }); +// The text of a stream whose chunks are all strings is one string of the sum of their lengths. +// The consumer joined them in a WTF::StringBuilder that aborts the process when it cannot grow: +// when the allocator refuses its buffer, and when it doubles the buffer for the first 16-bit +// chunk and the double is longer than a 16-bit string can be. +describe("the text of string chunks is one allocation of its length", () => { + const outOfMemory = "RangeError: Out of memory"; + // "aaabbc" is "a3 b2 c1". + const runs = `text => text.replace(/(.)\\1*/gs, (run, character) => character + run.length + " ").trim()`; + const consume = (chunks: string, text: string, describeText = runs) => ` + const megabyte = letter => Buffer.alloc(1024 * 1024, letter).toString("latin1"); + const chunks = ${chunks}; + const stream = new ReadableStream({ + start(controller) { + for (const chunk of chunks) controller.enqueue(chunk); + controller.close(); + }, + }); + const describeText = ${describeText}; + const settled = await ${text}.then(text => ({ text: describeText(text) }), e => ({ rejected: e.name + ": " + e.message })); + console.log(JSON.stringify(settled)); + `; + + // With Malloc=1 WebKit allocates through the system allocator, so ASAN's cap of 4 MiB for one + // allocation covers the text. ASAN logs every allocation that it refuses to stderr. + describe.skipIf(!isASAN)("under a cap of 4 MiB for one allocation", () => { + const MIB = 1024 * 1024; + const env = { + ...bunEnv, + Malloc: "1", + ASAN_OPTIONS: [bunEnv.ASAN_OPTIONS, "allocator_may_return_null=1", "max_allocation_size_mb=4", "detect_leaks=0"] + .filter(Boolean) + .join(":"), + }; + const megabytes = (letters: string) => `[...${JSON.stringify(letters)}].map(megabyte)`; + + test.concurrent.each([ + ["three megabytes", megabytes("abc"), { text: `a${MIB} b${MIB} c${MIB}` }], + ["four megabytes", megabytes("abcd"), { rejected: outOfMemory }], + // A megabyte of 16-bit characters is 2 MiB. + ["one megabyte, then a 16-bit character", `[megabyte("a"), "\\u20AC"]`, { text: `a${MIB} \u20AC1` }], + ["three megabytes, then a 16-bit character", `[...${megabytes("abc")}, "\\u20AC"]`, { rejected: outOfMemory }], + ["a 16-bit character, then three megabytes", `["\\u20AC", ...${megabytes("abc")}]`, { rejected: outOfMemory }], + ])("%s", async (_name, chunks, expected) => { + const { stdout, exitCode } = await run(consume(chunks, "Bun.readableStreamToText(stream)"), env); + expect({ stdout: JSON.parse(stdout || "null"), exitCode }).toEqual({ stdout: expected, exitCode: 0 }); + }); + + test.concurrent.each([ + "stream.text()", + "new Response(stream).text()", + "stream.json()", + "new Response(stream).json()", + ])("%s of four megabytes", async text => { + const { stdout, exitCode } = await run(consume(megabytes("abcd"), text), env); + expect({ stdout: JSON.parse(stdout || "null"), exitCode }).toEqual({ + stdout: { rejected: outOfMemory }, + exitCode: 0, + }); + }); + }); + + // This text fits in a string. It is 2 GiB in 16 bits, beside the gigabyte of the chunk. + test.skipIf(totalmem() < 12 * 1024 ** 3)( + "a 16-bit character after a gigabyte of Latin-1", + async () => { + const chunks = `[Buffer.alloc(2 ** 30, "x").toString("latin1"), "\\u20AC"]`; + const ends = `text => ({ length: text.length, start: text.slice(0, 3), end: text.slice(-3) })`; + const { stdout, stderr, exitCode } = await run(consume(chunks, "Bun.readableStreamToText(stream)", ends)); + expect({ stdout: JSON.parse(stdout || "null"), stderr, exitCode }).toEqual({ + stdout: { text: { length: 2 ** 30 + 1, start: "xxx", end: "xx\u20AC" } }, + stderr: "", + exitCode: 0, + }); + }, + 60_000, + ); +}); + // TextDecoderStream joins a chunk with the bytes it carried over from an incomplete UTF-8 // sequence. The join sized a WTF::Vector that aborts past 2^31-1 bytes, so a 2^31-byte chunk // after a carried byte killed the process before it read a byte. The child reserves the diff --git a/test/js/web/streams/streams.test.js b/test/js/web/streams/streams.test.js index ed0b09ce8e81..187a0b998198 100644 --- a/test/js/web/streams/streams.test.js +++ b/test/js/web/streams/streams.test.js @@ -1895,6 +1895,22 @@ describe("multi-chunk consumers produce exactly the concatenated bytes", () => { await expect(new Response(source([42])).text()).rejects.toThrow(expect.objectContaining({ name: "TypeError" })); }); + // The text of string chunks is one string of the sum of their lengths, 16-bit when a chunk is. + const unresolvedRopes = () => { + const part = Buffer.alloc(40, "p").toString(); + return [1, 2, 3].map(i => part + i + (i === 2 ? "\u20AC" : "\u00E9") + part); + }; + it.each([ + ["Latin-1 chunks", () => ["caf\u00E9", "", " au lait"], "caf\u00E9 au lait"], + ["a 16-bit chunk after Latin-1 chunks", () => ["abc", "d\u00E9", "\u20AC"], "abcd\u00E9\u20AC"], + ["a Latin-1 chunk after a 16-bit chunk", () => ["\u4F60\u597D", "abc"], "\u4F60\u597Dabc"], + ["only empty chunks", () => ["", ""], ""], + ["chunks that are ropes", unresolvedRopes, unresolvedRopes().join("")], + ])("text: string chunks join in order: %s", async (_name, chunks, text) => { + expect(await Bun.readableStreamToText(source(chunks()))).toBe(text); + expect(await new Response(source(chunks())).text()).toBe(text); + }); + it("a detached chunk throws", () => { const chunk = new Uint8Array([1, 2, 3]); structuredClone(chunk.buffer, { transfer: [chunk.buffer] }); From 375b4212d17f701a48068a46294670b3d8251413 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Wed, 30 Sep 2026 05:05:44 +0000 Subject: [PATCH 2/4] streams: a text result without its BOM shares the buffer of the text String::substring(1) copies the text. After the join of the string chunks that was a second allocation of the size of the text, from an allocator that aborts the process when it fails. substringSharingImpl(1) allocates the header of a substring. The gigabyte test joins 16 chunks of 64 MiB, so its child holds 2.2 GB like the other children of the file. It does not run in a debug build, which takes 6 s for the copy. --- .../webcore/streams/BunStreamConsumers.cpp | 6 +-- .../web/streams/streams-string-limit.test.ts | 49 +++++++++++++------ test/js/web/streams/streams.test.js | 4 ++ 3 files changed, 41 insertions(+), 18 deletions(-) diff --git a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp index 58ea99874c4e..c06504fd2ab0 100644 --- a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp +++ b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp @@ -232,7 +232,7 @@ using WebCore::JSStreamsRuntime; WTF::String withoutUTF8BOM(const WTF::String& string) { if (string.length() && string[0] == 0xFEFF) - return string.substring(1); + return string.substringSharingImpl(1); return string; } @@ -748,7 +748,7 @@ static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject* globalObje WTF::String rope = accumulator.rope.toString(); releaseAccumulated(); if (rope.length() && rope[0] == 0xFEFF) - return rope.substring(1); + return rope.substringSharingImpl(1); return rope; } // estimatedLength never overcounts the bytes, so an estimate past the limit is final. @@ -778,7 +778,7 @@ static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject* globalObje if (accumulator.rope.length()) { WTF::String rope = accumulator.rope.toString(); if (rope[0] == 0xFEFF) - rope = rope.substring(1); + rope = rope.substringSharingImpl(1); if (!appendUTF8WithinStringLimit(rope, bytes)) [[unlikely]] { releaseAccumulated(); throwOutOfMemoryError(globalObject, scope); diff --git a/test/js/web/streams/streams-string-limit.test.ts b/test/js/web/streams/streams-string-limit.test.ts index 643b9661fd76..2a92772f371d 100644 --- a/test/js/web/streams/streams-string-limit.test.ts +++ b/test/js/web/streams/streams-string-limit.test.ts @@ -1,5 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { bunEnv, bunExe, isASAN } from "harness"; +import { bunEnv, bunExe, emptyProcessMaxRSS, isASAN, isDebug, runFixtureMaxRSS } from "harness"; import { totalmem } from "node:os"; // Consuming a stream as text must reject with a catchable error when the accumulated @@ -248,21 +248,40 @@ describe("the text of string chunks is one allocation of its length", () => { }); }); - // This text fits in a string. It is 2 GiB in 16 bits, beside the gigabyte of the chunk. - test.skipIf(totalmem() < 12 * 1024 ** 3)( - "a 16-bit character after a gigabyte of Latin-1", - async () => { - const chunks = `[Buffer.alloc(2 ** 30, "x").toString("latin1"), "\\u20AC"]`; - const ends = `text => ({ length: text.length, start: text.slice(0, 3), end: text.slice(-3) })`; - const { stdout, stderr, exitCode } = await run(consume(chunks, "Bun.readableStreamToText(stream)", ends)); - expect({ stdout: JSON.parse(stdout || "null"), stderr, exitCode }).toEqual({ - stdout: { text: { length: 2 ** 30 + 1, start: "xxx", end: "xx\u20AC" } }, - stderr: "", - exitCode: 0, + // This text fits in a string: it is 2 GiB in 16 bits. A debug build takes 6 s to copy it. + test.skipIf(!enoughMemory || isDebug)("a 16-bit character after a gigabyte of Latin-1", async () => { + const chunks = `[...Array(16).fill(Buffer.alloc(2 ** 26, "x").toString("latin1")), "\\u20AC"]`; + const ends = `text => ({ length: text.length, start: text.slice(0, 3), end: text.slice(-3) })`; + const { stdout, stderr, exitCode } = await run(consume(chunks, "Bun.readableStreamToText(stream)", ends)); + expect({ stdout: JSON.parse(stdout || "null"), stderr, exitCode }).toEqual({ + stdout: { text: { length: 2 ** 30 + 1, start: "xxx", end: "xx\u20AC" } }, + stderr: "", + exitCode: 0, + }); + }); + + // The text without its BOM shares the buffer of the join. A copy is a second allocation of that size. + test("a BOM before 128 MiB of Latin-1", async () => { + const MIB = 1024 * 1024; + const fixture = ` + const megabyte = Buffer.alloc(${MIB}, "x").toString("latin1"); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue("\\uFEFF"); + for (let i = 0; i < 128; i++) controller.enqueue(megabyte); + controller.close(); + }, }); - }, - 60_000, - ); + const text = await stream.text(); + console.log(JSON.stringify({ length: text.length, start: text.slice(0, 3) })); + `; + const [peak, emptyPeak] = await Promise.all([ + runFixtureMaxRSS(fixture, { length: 128 * MIB, start: "xxx" }), + emptyProcessMaxRSS(), + ]); + // The join is 256 MiB in 16 bits, and a copy of it makes 512 MiB. + expect((peak - emptyPeak) / MIB).toBeLessThan(384); + }); }); // TextDecoderStream joins a chunk with the bytes it carried over from an incomplete UTF-8 diff --git a/test/js/web/streams/streams.test.js b/test/js/web/streams/streams.test.js index 187a0b998198..49dfa966f32b 100644 --- a/test/js/web/streams/streams.test.js +++ b/test/js/web/streams/streams.test.js @@ -1906,6 +1906,10 @@ describe("multi-chunk consumers produce exactly the concatenated bytes", () => { ["a Latin-1 chunk after a 16-bit chunk", () => ["\u4F60\u597D", "abc"], "\u4F60\u597Dabc"], ["only empty chunks", () => ["", ""], ""], ["chunks that are ropes", unresolvedRopes, unresolvedRopes().join("")], + ["a BOM before the text", () => ["\uFEFF", "abc"], "abc"], + ["two BOMs before the text", () => ["\uFEFF", "\uFEFFabc"], "abc"], + ["three BOMs before the text", () => ["\uFEFF\uFEFF", "\uFEFFabc"], "\uFEFFabc"], + ["only a BOM", () => ["\uFEFF", ""], ""], ])("text: string chunks join in order: %s", async (_name, chunks, text) => { expect(await Bun.readableStreamToText(source(chunks()))).toBe(text); expect(await new Response(source(chunks())).text()).toBe(text); From 59d03a08da8b2789f80576b01061eca013b5e611 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:27:38 +0000 Subject: [PATCH 3/4] streams: leave the BOM out in the join, and copy a rope chunk from its fibers The text without its BOM was a substring that shared the buffer of the join. A string of that kind is copied in full by each structuredClone() and postMessage(), from an allocator that aborts when it fails. The join now counts the U+FEFF code units at the start of the chunks, at most two, and leaves them out of its one allocation. The strips of the other arms are the copies of main again. JSString::resolveToBuffer() copies a chunk that is a rope from the strings of the rope. view() made the string of each rope first. A 16-bit text of 2,147,483,636 code units has a test that every build runs. allocationCapEnv() in the harness is the env of the tests that need an allocator that refuses. --- .../webcore/streams/BunStreamConsumers.cpp | 66 +++++++++++++---- test/harness.ts | 23 ++++++ .../web/streams/streams-string-limit.test.ts | 71 ++++++++++++------- test/js/web/streams/streams.test.js | 9 +++ 4 files changed, 132 insertions(+), 37 deletions(-) diff --git a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp index c06504fd2ab0..4eb29867e254 100644 --- a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp +++ b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp @@ -232,7 +232,7 @@ using WebCore::JSStreamsRuntime; WTF::String withoutUTF8BOM(const WTF::String& string) { if (string.length() && string[0] == 0xFEFF) - return string.substringSharingImpl(1); + return string.substring(1); return string; } @@ -540,20 +540,58 @@ static JSValue convertChunksToBytes(JSGlobalObject* globalObject, JSValue chunks static JSValue textAccumulatorWrite(JSC::VM& vm, JSGlobalObject*, JSC::JSObject* owner, BunTextAccumulator&, JSValue chunk); static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject*, JSC::JSObject* owner, BunTextAccumulator&); -// The string chunks as one string of `length` code units, from one allocation. Null when that allocation fails. +// The number of U+FEFF code units, at most two, at the start of the text of the string chunks. +// stripTextResultBOM() removes as many from a text. A Latin-1 chunk cannot start with one. +static unsigned leadingBOMCount(JSGlobalObject* globalObject, const MarkedArgumentBuffer& chunks) +{ + auto scope = DECLARE_THROW_SCOPE(getVM(globalObject)); + unsigned count = 0; + for (unsigned i = 0; i < chunks.size() && count < 2; i++) { + JSString* chunk = asString(chunks.at(i)); + if (!chunk->length()) + continue; + if (chunk->is8Bit()) + break; + auto view = chunk->view(globalObject); + RETURN_IF_EXCEPTION(scope, 0); + unsigned inChunk = 0; + while (inChunk < view->length() && count < 2 && view[inChunk] == 0xFEFF) { + inChunk++; + count++; + } + if (inChunk < view->length()) + break; + } + return count; +} + +// The string chunks as one string, without the first `skip` of their `length` code units. It is one +// allocation of that size, and null when that allocation fails. A chunk that is a rope is copied from +// its fibers: it is not resolved first. template -static WTF::String tryJoinStringChunks(JSGlobalObject* globalObject, const MarkedArgumentBuffer& chunks, unsigned length) +static WTF::String tryJoinStringChunks(JSGlobalObject* globalObject, const MarkedArgumentBuffer& chunks, unsigned length, unsigned skip) { auto scope = DECLARE_THROW_SCOPE(getVM(globalObject)); std::span characters; - WTF::String joined = WTF::String::tryCreateUninitialized(length, characters); + WTF::String joined = WTF::String::tryCreateUninitialized(length - skip, characters); if (joined.isNull()) [[unlikely]] return joined; for (unsigned i = 0; i < chunks.size(); i++) { - auto chunk = asString(chunks.at(i))->view(globalObject); - RETURN_IF_EXCEPTION(scope, {}); - chunk->getCharacters(characters); - characters = characters.subspan(chunk->length()); + JSString* chunk = asString(chunks.at(i)); + unsigned chunkLength = chunk->length(); + if (skip && chunkLength) [[unlikely]] { + unsigned skipped = std::min(skip, chunkLength); + skip -= skipped; + if (skipped < chunkLength) { + auto view = chunk->view(globalObject); + RETURN_IF_EXCEPTION(scope, {}); + view->substring(skipped).getCharacters(characters); + characters = characters.subspan(chunkLength - skipped); + } + continue; + } + chunk->resolveToBuffer(characters.first(chunkLength)); + characters = characters.subspan(chunkLength); } return joined; } @@ -625,15 +663,17 @@ static JSValue convertChunksToText(JSGlobalObject* globalObject, JSValue chunksV throwOutOfMemoryError(globalObject, scope); return {}; } + unsigned skip = all8Bit ? 0 : leadingBOMCount(globalObject, values); + RETURN_IF_EXCEPTION(scope, {}); WTF::String joined = all8Bit - ? tryJoinStringChunks(globalObject, values, codeUnits.value()) - : tryJoinStringChunks(globalObject, values, codeUnits.value()); + ? tryJoinStringChunks(globalObject, values, codeUnits.value(), skip) + : tryJoinStringChunks(globalObject, values, codeUnits.value(), skip); RETURN_IF_EXCEPTION(scope, {}); if (joined.isNull()) [[unlikely]] { throwOutOfMemoryError(globalObject, scope); return {}; } - RELEASE_AND_RETURN(scope, jsString(vm, stripTextResultBOM(joined))); + RELEASE_AND_RETURN(scope, jsString(vm, WTF::move(joined))); } // Mixed string/binary chunks: drive the shared accumulator so adjacent-string rope @@ -748,7 +788,7 @@ static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject* globalObje WTF::String rope = accumulator.rope.toString(); releaseAccumulated(); if (rope.length() && rope[0] == 0xFEFF) - return rope.substringSharingImpl(1); + return rope.substring(1); return rope; } // estimatedLength never overcounts the bytes, so an estimate past the limit is final. @@ -778,7 +818,7 @@ static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject* globalObje if (accumulator.rope.length()) { WTF::String rope = accumulator.rope.toString(); if (rope[0] == 0xFEFF) - rope = rope.substringSharingImpl(1); + rope = rope.substring(1); if (!appendUTF8WithinStringLimit(rope, bytes)) [[unlikely]] { releaseAccumulated(); throwOutOfMemoryError(globalObject, scope); diff --git a/test/harness.ts b/test/harness.ts index 60fffcbc865c..3e9a7f21c043 100644 --- a/test/harness.ts +++ b/test/harness.ts @@ -338,6 +338,29 @@ export async function runFixtureMaxRSS(fixture: string, expected: unknown) { return maxRSS; } +/** + * The env of a child whose allocator refuses one allocation of more than + * `megabytes` MiB. It needs an ASAN build: use it under `skipIf(!isASAN)`. + * `Malloc=1` is not optional. Without it WebKit takes the buffers of its + * strings from bmalloc, which the cap does not reach, and a test that expects + * a refusal passes or fails for another reason. ASAN logs every allocation + * that it refuses to stderr. + */ +export function allocationCapEnv(megabytes: number): typeof bunEnv { + return { + ...bunEnv, + Malloc: "1", + ASAN_OPTIONS: [ + bunEnv.ASAN_OPTIONS, + "allocator_may_return_null=1", + `max_allocation_size_mb=${megabytes}`, + "detect_leaks=0", + ] + .filter(Boolean) + .join(":"), + }; +} + /** * Runs `cmd` (a script that prints `{"deltaMiB": number}` as its last stdout * line) under bun with ASAN quarantine disabled, and asserts the delta is below diff --git a/test/js/web/streams/streams-string-limit.test.ts b/test/js/web/streams/streams-string-limit.test.ts index 2a92772f371d..ae7ce6146458 100644 --- a/test/js/web/streams/streams-string-limit.test.ts +++ b/test/js/web/streams/streams-string-limit.test.ts @@ -1,5 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { bunEnv, bunExe, emptyProcessMaxRSS, isASAN, isDebug, runFixtureMaxRSS } from "harness"; +import { allocationCapEnv, bunEnv, bunExe, emptyProcessMaxRSS, isASAN, isDebug, runFixtureMaxRSS } from "harness"; import { totalmem } from "node:os"; // Consuming a stream as text must reject with a catchable error when the accumulated @@ -192,6 +192,7 @@ test.skipIf(!enoughMemory)("arrayBuffer() and bytes() reject mixed chunks summin // when the allocator refuses its buffer, and when it doubles the buffer for the first 16-bit // chunk and the double is longer than a 16-bit string can be. describe("the text of string chunks is one allocation of its length", () => { + const MIB = 1024 * 1024; const outOfMemory = "RangeError: Out of memory"; // "aaabbc" is "a3 b2 c1". const runs = `text => text.replace(/(.)\\1*/gs, (run, character) => character + run.length + " ").trim()`; @@ -209,17 +210,8 @@ describe("the text of string chunks is one allocation of its length", () => { console.log(JSON.stringify(settled)); `; - // With Malloc=1 WebKit allocates through the system allocator, so ASAN's cap of 4 MiB for one - // allocation covers the text. ASAN logs every allocation that it refuses to stderr. describe.skipIf(!isASAN)("under a cap of 4 MiB for one allocation", () => { - const MIB = 1024 * 1024; - const env = { - ...bunEnv, - Malloc: "1", - ASAN_OPTIONS: [bunEnv.ASAN_OPTIONS, "allocator_may_return_null=1", "max_allocation_size_mb=4", "detect_leaks=0"] - .filter(Boolean) - .join(":"), - }; + const env = allocationCapEnv(4); const megabytes = (letters: string) => `[...${JSON.stringify(letters)}].map(megabyte)`; test.concurrent.each([ @@ -248,6 +240,18 @@ describe("the text of string chunks is one allocation of its length", () => { }); }); + // A 16-bit string holds at most 2,147,483,635 characters. The consumer refuses a longer text + // before it allocates, so the child stays small. + test("a 16-bit text of 2,147,483,636 characters", async () => { + const chunks = `["\\u20AC", ...Array(2047).fill(megabyte("x")), megabyte("x").slice(0, 2 ** 20 - 13)]`; + const { stdout, stderr, exitCode } = await run(consume(chunks, "Bun.readableStreamToText(stream)")); + expect({ stdout: JSON.parse(stdout || "null"), stderr, exitCode }).toEqual({ + stdout: { rejected: outOfMemory }, + stderr: "", + exitCode: 0, + }); + }); + // This text fits in a string: it is 2 GiB in 16 bits. A debug build takes 6 s to copy it. test.skipIf(!enoughMemory || isDebug)("a 16-bit character after a gigabyte of Latin-1", async () => { const chunks = `[...Array(16).fill(Buffer.alloc(2 ** 26, "x").toString("latin1")), "\\u20AC"]`; @@ -260,27 +264,46 @@ describe("the text of string chunks is one allocation of its length", () => { }); }); - // The text without its BOM shares the buffer of the join. A copy is a second allocation of that size. - test("a BOM before 128 MiB of Latin-1", async () => { - const MIB = 1024 * 1024; + // The peak RSS of a child that reads a stream of string chunks as text, above the peak of an + // empty child, in MiB. + const peakOfText = async (chunks: string, report: string, expected: unknown) => { const fixture = ` - const megabyte = Buffer.alloc(${MIB}, "x").toString("latin1"); + const latin1 = (length, letter) => Buffer.alloc(length, letter).toString("latin1"); + ${chunks} const stream = new ReadableStream({ start(controller) { - controller.enqueue("\\uFEFF"); - for (let i = 0; i < 128; i++) controller.enqueue(megabyte); + for (const chunk of chunks) controller.enqueue(chunk); controller.close(); }, }); const text = await stream.text(); - console.log(JSON.stringify({ length: text.length, start: text.slice(0, 3) })); + console.log(JSON.stringify(${report})); `; - const [peak, emptyPeak] = await Promise.all([ - runFixtureMaxRSS(fixture, { length: 128 * MIB, start: "xxx" }), - emptyProcessMaxRSS(), - ]); - // The join is 256 MiB in 16 bits, and a copy of it makes 512 MiB. - expect((peak - emptyPeak) / MIB).toBeLessThan(384); + const [peak, emptyPeak] = await Promise.all([runFixtureMaxRSS(fixture, expected), emptyProcessMaxRSS()]); + return (peak - emptyPeak) / MIB; + }; + + // The join leaves the BOM out. No copy removes it, and the text is not a part of another + // string, so structuredClone() shares it. The text is 256 MiB in 16 bits. A copy makes 512 MiB. + test("a BOM before 128 MiB of Latin-1", async () => { + const peak = await peakOfText( + `const chunks = ["\\uFEFF", ...Array(128).fill(latin1(${MIB}, "x"))];`, + `{ length: text.length, start: text.slice(0, 3), clone: structuredClone(text).length }`, + { length: 128 * MIB, start: "xxx", clone: 128 * MIB }, + ); + expect(peak).toBeLessThan(384); + }); + + // The join copies a chunk that is a rope from the two strings of the rope. It does not make + // the string of each rope first, which is 128 MiB more. + test("128 MiB of chunks that are ropes", async () => { + const peak = await peakOfText( + `const a = latin1(${MIB / 2}, "a"), b = latin1(${MIB / 2}, "b"); + const chunks = Array.from({ length: 128 }, () => a + b);`, + `{ length: text.length, start: text.slice(0, 3), end: text.slice(-3) }`, + { length: 128 * MIB, start: "aaa", end: "bbb" }, + ); + expect(peak).toBeLessThan(192); }); }); diff --git a/test/js/web/streams/streams.test.js b/test/js/web/streams/streams.test.js index 49dfa966f32b..c6859c306f9b 100644 --- a/test/js/web/streams/streams.test.js +++ b/test/js/web/streams/streams.test.js @@ -1910,6 +1910,15 @@ describe("multi-chunk consumers produce exactly the concatenated bytes", () => { ["two BOMs before the text", () => ["\uFEFF", "\uFEFFabc"], "abc"], ["three BOMs before the text", () => ["\uFEFF\uFEFF", "\uFEFFabc"], "\uFEFFabc"], ["only a BOM", () => ["\uFEFF", ""], ""], + ["two BOMs and text in one chunk", () => ["\uFEFF\uFEFFab", "c"], "abc"], + ["BOMs with empty chunks between them", () => ["", "\uFEFF", "", "\uFEFF", "abc"], "abc"], + ["a BOM after the first character", () => ["a", "\uFEFFbc"], "a\uFEFFbc"], + ["a BOM at the start of a rope", () => ["\uFEFF" + unresolvedRopes()[1], "abc"], unresolvedRopes()[1] + "abc"], + [ + "a BOM at the start of a part of another string", + () => [("x\uFEFF" + unresolvedRopes()[0]).slice(1), "abc"], + unresolvedRopes()[0] + "abc", + ], ])("text: string chunks join in order: %s", async (_name, chunks, text) => { expect(await Bun.readableStreamToText(source(chunks()))).toBe(text); expect(await new Response(source(chunks())).text()).toBe(text); From 788496a18a443b615a4b60d8f3b3b372407aa018 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Thu, 1 Oct 2026 22:32:03 +0000 Subject: [PATCH 4/4] streams: one line for each comment of the join --- src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp index 4eb29867e254..2d6081fc688a 100644 --- a/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp +++ b/src/jsc/bindings/webcore/streams/BunStreamConsumers.cpp @@ -540,8 +540,7 @@ static JSValue convertChunksToBytes(JSGlobalObject* globalObject, JSValue chunks static JSValue textAccumulatorWrite(JSC::VM& vm, JSGlobalObject*, JSC::JSObject* owner, BunTextAccumulator&, JSValue chunk); static WTF::String finishTextAccumulator(JSC::VM& vm, JSGlobalObject*, JSC::JSObject* owner, BunTextAccumulator&); -// The number of U+FEFF code units, at most two, at the start of the text of the string chunks. -// stripTextResultBOM() removes as many from a text. A Latin-1 chunk cannot start with one. +// How many U+FEFF code units, at most two, start the text of the string chunks. The text result leaves them out. static unsigned leadingBOMCount(JSGlobalObject* globalObject, const MarkedArgumentBuffer& chunks) { auto scope = DECLARE_THROW_SCOPE(getVM(globalObject)); @@ -565,9 +564,7 @@ static unsigned leadingBOMCount(JSGlobalObject* globalObject, const MarkedArgume return count; } -// The string chunks as one string, without the first `skip` of their `length` code units. It is one -// allocation of that size, and null when that allocation fails. A chunk that is a rope is copied from -// its fibers: it is not resolved first. +// The string chunks as one string without its first `skip` code units, from one allocation. Null when that fails. template static WTF::String tryJoinStringChunks(JSGlobalObject* globalObject, const MarkedArgumentBuffer& chunks, unsigned length, unsigned skip) { @@ -590,6 +587,7 @@ static WTF::String tryJoinStringChunks(JSGlobalObject* globalObject, const Marke } continue; } + // This copies a rope from its fibers. It does not make the string of the rope. chunk->resolveToBuffer(characters.first(chunkLength)); characters = characters.subspan(chunkLength); }