Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 5 additions & 35 deletions src/js_printer/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -4284,41 +4284,11 @@
self.print(b" ");
}

if IS_BUN_PLATFORM {
// Translate any non-ASCII to unicode escape sequences
let mut ascii_start: usize = 0;
let mut is_ascii = false;
let iter = CodepointIterator::init(&e.value);
let mut cursor = strings::Cursor::default();
while iter.next(&mut cursor) {
match cursor.c as u32 {
FIRST_ASCII..=LAST_ASCII => {
if !is_ascii {
ascii_start = cursor.i as usize;
is_ascii = true;
}
}
_ => {
if is_ascii {
self.print(&e.value[ascii_start..(cursor.i as usize)]);
is_ascii = false;
}

match cursor.c as u32 {
c @ 0..=0xFFFF => self.print(&bmp_escape(c)[..]),
c => self.print(&surrogate_pair_escape(c)[..]),
}
}
}
}

if is_ascii {
self.print(&e.value[ascii_start..]);
}
} else {
// UTF8 sequence is fine
self.print(&e.value[..]);
}
// The pattern is printed verbatim (UTF-8), even under `IS_BUN_PLATFORM`:
// rewriting `/¶/u` as `/\u00B6/u` changes `RegExp.prototype.source` at
// runtime. The consumers of the transpiled buffer treat it as UTF-8
// (see `String::clone_utf8` at the `ResolvedSource` construction sites).
self.print(&e.value[..]);

Check failure on line 4291 in src/js_printer/lib.rs

View check run for this annotation

Claude / Claude Code Review

Removing IS_BUN_PLATFORM regex escape breaks Latin-1 consumers of bundler output (already_bundled fast path + bytecode generation)

Removing the `IS_BUN_PLATFORM` regex escape breaks the all-ASCII invariant that two other consumers of `--target=bun` printer output still depend on: the `already_bundled` fast path (`String::clone_latin1(&parse_result.source.contents)` in `src/runtime/jsc_hooks.rs:2759` and the same call in `RuntimeTranspilerStore.rs`) and the build-time bytecode generator (`generateCachedModuleByteCodeFromSourceCode` / the CJS twin in `src/jsc/bindings/ZigSourceProvider.cpp:211-214,246-250`, which construct `W
Comment on lines +4287 to +4291

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🔴 Removing the IS_BUN_PLATFORM regex escape breaks the all-ASCII invariant that two other consumers of --target=bun printer output still depend on: the already_bundled fast path (String::clone_latin1(&parse_result.source.contents) in src/runtime/jsc_hooks.rs:2759 and the same call in RuntimeTranspilerStore.rs) and the build-time bytecode generator (generateCachedModuleByteCodeFromSourceCode / the CJS twin in src/jsc/bindings/ZigSourceProvider.cpp:211-214,246-250, which construct WTF::String(std::span<const Latin1Character>(...))). A source containing /¶/u now round-trips through bun build --target=bun → bun run, or bun build --bytecode / --compile, as /¶/u — the regex no longer matches, which is a functional regression (previously it emitted /\u00B6/u and matched correctly). These are sibling sites of the four clone_latin1 → clone_utf8 conversions this PR does apply and need the same treatment (or the bundler path should keep escaping).

Extended reasoning...

What the bug is

This PR deletes the IS_BUN_PLATFORM non-ASCII-escaping branch in print_reg_exp_literal so regex patterns are printed verbatim as UTF-8, and updates the runtime-transpiler consumers of that buffer to decode UTF-8 (clone_latin1 → clone_utf8 at four sites, plus an is_all_ascii fallback in ref_counted_resolved_source). But two other consumers of the same printer's output still interpret it as Latin-1 and were not updated:

  1. The already_bundled runtime fast path — bun_core::String::clone_latin1(&source.contents) at src/runtime/jsc_hooks.rs:2759 and the identical block in src/jsc/RuntimeTranspilerStore.rs (the AlreadyBundled arm around line 1005).
  2. The build-time bytecode generator — generateCachedModuleByteCodeFromSourceCode and generateCachedCommonJSProgramByteCodeFromSourceCode in src/jsc/bindings/ZigSourceProvider.cpp:211-214 / :246-250 declare const Latin1Character* inputSourceCode and construct WTF::String(std::span<const Latin1Character>(inputSourceCode, inputSourceCodeSize)).

Both paths feed on the bundler's --target=bun output, which after this change can contain raw multi-byte UTF-8.

The specific code path

For the bundler round-trip: bun build --target=bun input.js → LinkerContext.rs calls js_printer::print_with_writer(..., ast.target, ...) → lib.rs:7816 sees target.is_bun() and dispatches to print_with_writer_and_platform::<_, /*IS_BUN_PLATFORM=*/true, _> (the type alias at lib.rs:7863-7864 sets both ASCII_ONLY=true and IS_BUN_PLATFORM=true). Strings and identifiers are still ASCII-escaped by ASCII_ONLY, but after this PR print_reg_exp_literal writes the pattern verbatim, so /¶/u puts raw bytes 0xC2 0xB6 into the chunk. postProcessJSChunk.rs prepends // @bun and the bundle is written as UTF-8. When the bundle is executed, the // @bun pragma triggers AlreadyBundled, and both jsc_hooks.rs:2759 and the async twin in RuntimeTranspilerStore.rs build the ResolvedSource with String::clone_latin1(&source.contents) → BunString__fromLatin1, which memcpy's each byte as one Latin-1 code point.

For bytecode: --bytecode and --compile both force target=bun (Arguments.rs / build_command.rs), so the same IS_BUN_PLATFORM=true printer runs. generateChunksInParallel.rs:1057 / writeOutputFilesToDisk.rs:412 pass &code_result.buffer to generate_cached_bytecode → __bun_jsc_generate_cached_bytecode → the C++ FFI, with no transcoding. The C++ side then constructs WTF::String from a Latin1Character span — bytewise Latin-1 decode.

Why existing code doesn't prevent it

The four clone_latin1 → clone_utf8 conversions in this PR sit on the runtime-transpiler print path (freshly-transpiled source going straight to JSC). The two sites above sit on different paths that consume the bundler's --target=bun output — the // @bun fast path skips transpilation entirely and hands the raw file bytes to JSC, and bytecode generation happens at build time in C++ before any of the Rust-side runtime consumers are involved. Fixing the four runtime sites doesn't touch either of these; fixing the already_bundled sites doesn't fix the C++ bytecode path and vice versa. Before this PR the IS_BUN_PLATFORM regex escape guaranteed --target=bun output was all-ASCII, so clone_latin1 / the Latin1Character* span were equivalent to UTF-8 decoding; the PR removes that guarantee without updating these consumers.

Step-by-step proof

Take input.js = console.log(/¶/u.test("¶")).

Bundle round-trip:

  1. bun build --target=bun input.js -o out.js → printer emits /¶/u verbatim; out.js contains bytes … 2F C2 B6 2F 75 … prefixed with // @bun.
  2. bun out.js → parser sees // @bun, returns AlreadyBundled.
  3. jsc_hooks.rs:2759 calls String::clone_latin1(&source.contents) → BunString__fromLatin1 treats 0xC2 as U+00C2 and 0xB6 as U+00B6.
  4. JSC receives console.log(/¶/u.test("¶")) (the string literal was already ASCII-escaped by ASCII_ONLY, so it decodes fine); prints false.
  5. Before this PR the bundle contained /\u00B6/u, which clone_latin1 passed through byte-for-byte, and JSC parsed as a regex matching U+00B6 → printed true. So this is a functional regression in matching behaviour, not just .source cosmetics.

Bytecode / compile:

  1. bun build --compile input.js -o app (or --target=bun --bytecode) → same printer, same 0xC2 0xB6 in code_result.buffer.
  2. generate_cached_bytecode forwards buffer.as_ptr()/len() unchanged to generateCachedModuleByteCodeFromSourceCode.
  3. ZigSourceProvider.cpp:213 builds WTF::String(std::span<const Latin1Character>(ptr, len)) → JSC parses /¶/u and bakes it into the .jsc cache / compiled executable.
  4. Running ./app prints false; before this PR it printed true.

Impact

bun build --target=bun → bun run, bun build --bytecode, and bun build --compile all produce silently-wrong runtime behaviour for any source containing a non-ASCII regex literal — the regex no longer matches what it did before, and .source is mojibake. This reintroduces (in a worse form — matching is broken, not just .source) the exact bug class this PR fixes for the runtime transpiler, on the bundler / compiled-binary path. Per REVIEW.md "Fix the whole class in the same PR", these are missed sibling sites of the sweep the PR performs.

How to fix

Either:

  • Convert the two already_bundled clone_latin1 calls to clone_utf8 (matching the four this PR already converts), and change ZigSourceProvider.cpp's two bytecode entry points to construct the WTF::String via WTF::String::fromUTF8(...) (or check charactersAreAllASCII and branch); or
  • Keep the \uXXXX escaping for regex literals on the bundler path only (gate on something other than IS_BUN_PLATFORM, since IS_BUN_PLATFORM is shared with the runtime transpiler), leaving the runtime path verbatim.

Whichever is chosen, the tests should cover the bun build --target=bun → bun run round-trip and --compile / --bytecode (per REVIEW.md "Cover the variant matrix … alternate modes (--compile, --bytecode)").


// Need a space before the next identifier to avoid it turning into flags
self.prev_reg_exp_end = self.writer.written();
Expand Down
5 changes: 4 additions & 1 deletion src/jsc/AsyncModule.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1375,7 +1375,10 @@ impl AsyncModule {
}

Ok(ResolvedSource {
source_code: BunString::clone_latin1(printer.ctx.get_written()),
// `clone_utf8`: RegExp literals are printed verbatim (non-ASCII
// bytes possible); `BunString__fromBytes` stays Latin-1 when the
// output is all-ASCII.
source_code: BunString::clone_utf8(printer.ctx.get_written()),
specifier: BunString::init(specifier),
source_url: BunString::init(path.text),
is_commonjs_module,
Expand Down
23 changes: 16 additions & 7 deletions src/jsc/RuntimeTranspilerCache.rs
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,10 @@
/// path reinstates the bug for any previously-cached TLA module (#30887).
/// Version 23: `jsx.runtime`/`jsx.development` participate in the features hash,
/// and tsconfig `"jsx": "react-jsx"` now emits the production runtime (#4227).
const EXPECTED_VERSION: u32 = 23;
/// Version 24: RegExp literals are printed verbatim (no `\uXXXX` escaping of
/// non-ASCII) so `RegExp.prototype.source` matches the source text (#13853);
/// cached output is now tagged `Encoding::UTF8`.
const EXPECTED_VERSION: u32 = 24;

/// Source files smaller than this are not written to / read from the on-disk
/// transpiler cache. Originally 50 KiB, which excluded almost every file in a
Expand Down Expand Up @@ -1009,7 +1012,10 @@
return;
}
debug_assert!(self.entry.is_none());
let output_code = BunString::clone_latin1(output_code_bytes);
// Printer output is ASCII except for RegExp literals (printed verbatim
// so `.source` is preserved); `clone_utf8` keeps the Latin-1 fast path
// for the common all-ASCII case and transcodes to UTF-16 otherwise.
let output_code = BunString::clone_utf8(output_code_bytes);
// Refcount stays at 1, sole owner.
// BunString is Copy with no Drop, so an extra dupe_ref here would leak.
self.output_code = Some(output_code);
Expand All @@ -1028,7 +1034,7 @@
}
#[cfg(debug_assertions)]
{
bun_core::scoped_log!(cache, "put() = {} bytes", output_code.latin1().len());
bun_core::scoped_log!(cache, "put() = {} bytes", output_code_bytes.len());
}
}
}
Expand Down Expand Up @@ -1078,10 +1084,13 @@
}
debug_assert!(this.entry.is_none());

// Borrowed Latin-1 view: `to_file` only reads `byte_slice()` + the encoding
// tag (unmarked 8-bit ZigString -> Encoding::LATIN1, same as clone_latin1),
// and `output_code_bytes` outlives the synchronous `to_file` call.
let output_code = BunString::ascii(output_code_bytes);
// Borrowed UTF-8 view: `to_file` only reads `byte_slice()` + the
// encoding tag, and `output_code_bytes` outlives the synchronous
// `to_file` call. Printer output is ASCII except for RegExp
// literals (printed verbatim so `.source` is preserved), so tag as
// UTF-8; the read path's `Encoding::UTF8` branch handles both the
// all-ASCII fast path and the rare multi-byte case.
let output_code = BunString::borrow_utf8(output_code_bytes);

Check warning on line 1093 in src/jsc/RuntimeTranspilerCache.rs

View check run for this annotation

Claude / Claude Code Review

borrow_utf8 in vtable put routes every cache write through the Box::from memcpy arm in to_file

Switching `BunString::ascii` → `BunString::borrow_utf8` here now routes every cache write through `to_file`'s `if source_code.is_utf8() { OutputCode::Utf8(Box::from(...)) }` arm — a full alloc+memcpy of the printer output — where it previously hit the zero-copy `OutputCode::String(*source_code)` arm. `Entry::save` already handles a UTF-8-tagged `OutputCode::String` correctly (its match checks `str.is_utf8()` → `Encoding::UTF8`), so dropping the `is_utf8()` special case in `to_file` restores zero
Comment on lines +1087 to +1093

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 Switching BunString::ascii → BunString::borrow_utf8 here now routes every cache write through to_file's if source_code.is_utf8() { OutputCode::Utf8(Box::from(...)) } arm — a full alloc+memcpy of the printer output — where it previously hit the zero-copy OutputCode::String(*source_code) arm. Entry::save already handles a UTF-8-tagged OutputCode::String correctly (its match checks str.is_utf8() → Encoding::UTF8), so dropping the is_utf8() special case in to_file restores zero-copy with identical on-disk output. Perf-only on the cold-cache-write path, but it makes the "Borrowed UTF-8 view: to_file only reads byte_slice() + the encoding tag" comment above factually wrong.

Extended reasoning...

What changed and what it does

The vtable put bridge for the runtime transpiler cache changed the borrowed source-code view from BunString::ascii(output_code_bytes) to BunString::borrow_utf8(output_code_bytes) (RuntimeTranspilerCache.rs:1093). The updated comment frames this as a zero-copy borrow: "Borrowed UTF-8 view: to_file only reads byte_slice() + the encoding tag". That claim is now false.

The code path

BunString::borrow_utf8 (bun_core/string/mod.rs:185) → ZigString::init_utf8 (mod.rs:1383-1386) → mark_utf8() sets the UTF-8 ptr-tag on a Tag::ZigString. String::is_utf8() (mod.rs:543-545) is matches!(tag, ZigString|StaticZigString) && self.as_zig().is_utf8(), so it now returns true for this borrowed view.

to_file then branches:

let output_code: OutputCode = if source_code.is_utf8() {
    OutputCode::Utf8(Box::from(source_code.byte_slice()))   // full memcpy
} else {
    OutputCode::String(*source_code)                          // zero-copy borrow
};

Before this PR, BunString::ascii (mod.rs:193) → ZigString::init produced an unmarked ZigString → is_utf8() == false → the zero-copy OutputCode::String arm was taken. After this PR, every call takes the Box::from arm and heap-allocates + memcpys the entire printed output. That branch even carries a pre-existing // PERF: add a borrowed OutputCode variant to avoid the copy TODO — before this PR that TODO was dormant because nothing on the hot path hit it; this PR routes every cache write through it.

Step-by-step proof

  1. js_printer/lib.rs:7715 calls cache.put(printer.writer.slice(), ...) on the bun_ast::RuntimeTranspilerCache after printing, say, a 30 KB node_modules file.
  2. r#impl = Some(Jsc) (set in both RuntimeTranspilerStore.rs and jsc_hooks.rs), so put dispatches to this vtable arm.
  3. BunString::borrow_utf8(output_code_bytes) produces a Tag::ZigString with the UTF-8 ptr-tag set.
  4. to_file receives &output_code; source_code.is_utf8() returns true.
  5. Box::from(source_code.byte_slice()) allocates 30 KB on the worker's mimalloc heap and memcpys the printer buffer into it.
  6. Entry::save reads output_code.byte_slice() off the Box and writes it via pwritev.
  7. The Box is dropped at the end of to_file.

Contrast with the pre-PR path: step 3 produced an unmarked ZigString, step 4 returned false, step 5 was OutputCode::String(*source_code) (a 24-byte struct copy of a borrowed view), and step 6 read the same bytes directly from the printer buffer. Same on-disk output, zero extra allocation.

Note that the non-vtable RuntimeTranspilerCache::put (which uses clone_utf8 → a WTFStringImpl-tagged string) is unaffected — is_utf8() only returns true for ZigString/StaticZigString tags, so that path still hits the zero-copy arm. Only the vtable bridge regressed, but the vtable bridge is what every runtime transpile actually uses.

Why the copy is unnecessary

Entry::save already handles a UTF-8-tagged OutputCode::String correctly. Its output_encoding match arm for OutputCode::String(str) checks str.is_utf8() → Encoding::UTF8, and OutputCode::byte_slice() for the String variant returns the borrowed bytes via s.byte_slice(). So passing OutputCode::String(*source_code) unconditionally would produce identical on-disk output (same Encoding::UTF8 header byte, same body bytes) with no allocation.

Impact

Fires on every cache-miss transpile of a source file above the 4 KiB MINIMUM_CACHE_SIZE, on both the sync (jsc_hooks) and async (RuntimeTranspilerStore) paths. For a cold node_modules scan of ~1500 files at ~30 KB output each, that is ~45 MB of extra alloc+memcpy on the transpiler worker threads. This is on a path that is already I/O-bound (open/preallocate/pwritev), so wall-clock impact is small in absolute terms — hence nit severity — but it directly contradicts the code comment this PR added.

Suggested fix

Drop the is_utf8() special case in to_file and always take OutputCode::String(*source_code):

// `OutputCode::String` is a refcount-neutral by-value borrow (`BunString` is
// `Copy`, no `Drop`); `Entry::save` derives `Encoding::UTF8` from
// `str.is_utf8()` and reads `byte_slice()`, so no owning copy is needed.
let output_code = OutputCode::String(*source_code);

Alternatively, gate the Box::from arm on tag == WTFStringImpl if some other caller genuinely needs the owning box (none currently does).

let result = RuntimeTranspilerCache::to_file(
this.input_byte_length.unwrap(),
this.input_hash.unwrap(),
Expand Down
6 changes: 5 additions & 1 deletion src/jsc/RuntimeTranspilerStore.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1149,7 +1149,11 @@ impl TranspilerJob {
// `cache.output_code` (only the `r#impl == None` fallback does,
// and `r#impl` is `Some(Jsc)` here), so it is always `None`.
debug_assert!(cache.output_code.is_none());
let result = String::clone_latin1(written);
// `clone_utf8`: the printer emits ASCII-only output except for
// RegExp literals, which are printed verbatim so their `.source`
// is preserved; `BunString__fromBytes` keeps the Latin-1 fast
// path for the common all-ASCII case.
let result = String::clone_utf8(written);

// SAFETY: leaf scalar field read on `*vm`; see `vm` note above.
if written.len() > 1024 * 1024 * 2 || unsafe { (*vm).smol } {
Expand Down
12 changes: 12 additions & 0 deletions src/jsc/VirtualMachine.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3744,6 +3744,18 @@ impl VirtualMachine {
..Default::default()
};
}
// The ref-string cache wraps `code` in a Latin-1 external string. When
// the printer emitted non-ASCII UTF-8 (currently only RegExp literals,
// printed verbatim so `.source` is preserved), interning as Latin-1
// would corrupt those bytes, so fall back to a plain UTF-8 copy.
if !bun_core::strings::is_all_ascii(code) {
return ResolvedSource {
source_code: bun_core::String::clone_utf8(code),
specifier,
source_url: create_if_different(&specifier, source_url),
..Default::default()
};
}
// Const-generic bool can't be `!ADD_DOUBLE_REF`, so branch.
let source = if ADD_DOUBLE_REF {
self.ref_counted_string::<false>(code, hash_)
Expand Down
6 changes: 5 additions & 1 deletion src/runtime/jsc_hooks.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3154,7 +3154,11 @@ fn transpile_source_code_inner(
// `None`.
debug_assert!(cache.output_code.is_none());
let written_len = written.len();
let source_code = bun_core::String::clone_latin1(written);
// `clone_utf8`: the printer emits ASCII-only output except for
// RegExp literals, which are printed verbatim so their `.source`
// is preserved; `BunString__fromBytes` keeps the Latin-1 fast
// path for the common all-ASCII case.
let source_code = bun_core::String::clone_utf8(written);
// `printer.ctx.buffer.deinit()`: release the
// large/--smol print buffer now instead of holding it until the
// next transpile. Replacing the printer drops the old buffer
Expand Down
149 changes: 149 additions & 0 deletions test/regression/issue/13853/13853.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,149 @@
// https://github.com/oven-sh/bun/issues/13853
// RegExp literal .source must preserve non-ASCII characters from the original
// source text. Bun's runtime transpiler used to rewrite /¶/u as /\u00B6/u
// (to keep the printed output ASCII-only for a Latin-1 source pipeline),
// which changed the observable value of RegExp.prototype.source and broke
// packages such as parsel-js/Puppeteer that string-replace on .source.
import { expect, test } from "bun:test";
import { bunEnv, bunExe, tempDir } from "harness";
import { join } from "node:path";

test("RegExp literal .source preserves non-ASCII characters (#13853)", async () => {
// Spawn a fresh process so the fixture is run through the runtime transpiler
// (this test file itself is also transpiled, but the fixture's bytes are
// what we want the assertion to observe).
using dir = tempDir("issue-13853", {
"index.js": `
const results = {
latin1_no_u: /\u00b6/.source,
latin1_u: /\u00b6/u.source,
latin1_v: /\u00b6/v.source,
cjk: /\u8981\u66ff\u6362/u.source,
astral: /\u{1d54f}/u.source,
// via new RegExp the source text is already a runtime string,
// so this was never broken; kept as a sanity check
runtime: new RegExp("\u00b6", "u").source,
};
process.stdout.write(JSON.stringify(results));
`,
});

await using proc = Bun.spawn({
cmd: [bunExe(), "run", join(String(dir), "index.js")],
env: bunEnv,
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);

expect(stderr).toBe("");
const got = JSON.parse(stdout);
expect(got).toEqual({
latin1_no_u: "\u00b6",
latin1_u: "\u00b6",
latin1_v: "\u00b6",
cjk: "\u8981\u66ff\u6362",
astral: "\u{1d54f}",
runtime: "\u00b6",
});
expect(exitCode).toBe(0);
});

test("parsel-js .source.replace pattern works (#13853)", async () => {
// Minimal reduction of what Puppeteer's bundled parsel-js does for
// ::-p-xpath() / ::-p-text(): build a RegExp with a literal PILCROW SIGN
// placeholder, then .source.replace("\u00b6*", ".*") to derive a second
// pattern. If .source escaped \u00b6 to "\\u00B6" the replace would miss
// and the derived pattern would fail to capture the argument.
using dir = tempDir("issue-13853-parsel", {
"index.js": `
const TOKEN = /::(?<name>[-\\w]+)(?:\\((?<argument>\u00b6*)\\))?/gu;
const src = TOKEN.source.replace("(?<argument>\u00b6*)", "(?<argument>.*)");
const derived = new RegExp(src, "gu");
derived.lastIndex = 0;
const m = derived.exec("::-p-xpath(//div)");
process.stdout.write(JSON.stringify(m && m.groups));
`,
});

await using proc = Bun.spawn({
cmd: [bunExe(), "run", join(String(dir), "index.js")],
env: bunEnv,
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);

expect(stderr).toBe("");
expect(JSON.parse(stdout)).toEqual({ name: "-p-xpath", argument: "//div" });
expect(exitCode).toBe(0);
});

test("transpiler cache round-trip preserves non-ASCII RegExp .source (#13853)", async () => {
// The on-disk transpiler cache used to tag printer output as Latin-1, so a
// non-ASCII RegExp literal would be corrupted when read back on a cache hit.
// The file must exceed the 4 KiB minimum cache size.
const pad = Buffer.alloc(8 * 1024, "a").toString();
using dir = tempDir("issue-13853-cache", {
"a.js": `/* ${pad} */\nprocess.stdout.write(/\u00b6\u65e5/u.source);\n`,
});
const cacheDir = join(String(dir), ".cache");
const env = {
...bunEnv,
BUN_RUNTIME_TRANSPILER_CACHE_PATH: cacheDir,
BUN_DEBUG_ENABLE_RESTORE_FROM_TRANSPILER_CACHE: "1",
};

// First run: cache miss, writes the cache entry.
{
await using proc = Bun.spawn({
cmd: [bunExe(), "run", join(String(dir), "a.js")],
env,
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);
expect(stderr).toBe("");
expect(stdout).toBe("\u00b6\u65e5");
expect(exitCode).toBe(0);
}

// Second run: cache hit, reads the entry back.
{
await using proc = Bun.spawn({
cmd: [bunExe(), "run", join(String(dir), "a.js")],
env,
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);
expect(stderr).toBe("");
expect(stdout).toBe("\u00b6\u65e5");
expect(exitCode).toBe(0);
}
});

test("non-ASCII RegExp literal still matches correctly (#2005 stays fixed)", async () => {
using dir = tempDir("issue-13853-match", {
"index.js": `
const text = "\u8fd9\u662f\u4e00\u6bb5\u8981\u66ff\u6362\u7684\u6587\u5b57";
process.stdout.write(JSON.stringify({
literal: text.replace(/\u8981\u66ff\u6362/, ""),
ctor: text.replace(new RegExp("\u8981\u66ff\u6362"), ""),
}));
`,
});
await using proc = Bun.spawn({
cmd: [bunExe(), "run", join(String(dir), "index.js")],
env: bunEnv,
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);
expect(stderr).toBe("");
expect(JSON.parse(stdout)).toEqual({
literal: "\u8fd9\u662f\u4e00\u6bb5\u7684\u6587\u5b57",
ctor: "\u8fd9\u662f\u4e00\u6bb5\u7684\u6587\u5b57",
});
expect(exitCode).toBe(0);
});
Loading